{ "requesty": { "id": "requesty", "env": ["REQUESTY_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://router.requesty.ai/v1", "name": "Requesty", "doc": "https://requesty.ai/solution/llm-routing/models", "models": { "xai/grok-4": { "id": "xai/grok-4", "name": "Grok 4", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-09", "last_updated": "2025-09-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75, "cache_write": 3 } }, "xai/grok-4-fast": { "id": "xai/grok-4-fast", "name": "Grok 4 Fast", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 64000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05, "cache_write": 0.2 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.31, "cache_write": 2.375, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075, "cache_write": 0.55 } }, "google/gemini-3-pro-preview": { "id": "google/gemini-3-pro-preview", "name": "Gemini 3 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "cache_write": 4.5 } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 1 } }, "openai/gpt-5.2-chat": { "id": "openai/gpt-5.2-chat", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "GPT-5.2 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "audio", "image", "video"], "output": ["text", "audio", "image"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "openai/gpt-5-chat": { "id": "openai/gpt-5-chat", "name": "GPT-5 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4 Mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.28 } }, "openai/gpt-5.1-chat": { "id": "openai/gpt-5.1-chat", "name": "GPT-5.1 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT-5.1-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.1-codex-max": { "id": "openai/gpt-5.1-codex-max", "name": "GPT-5.1-Codex-Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.1, "output": 9, "cache_read": 0.11 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT-5.3-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "GPT-5.1-Codex-Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 100000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-5-image": { "id": "openai/gpt-5-image", "name": "GPT-5 Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10-01", "release_date": "2025-10-14", "last_updated": "2025-10-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 5, "output": 10, "cache_read": 1.25 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.03 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16000, "output": 4000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.01 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "cache_read": 30 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.08 } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "GPT-5 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10-01", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT-5.2-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "anthropic/claude-3-7-sonnet": { "id": "anthropic/claude-3-7-sonnet", "name": "Claude Sonnet 3.7", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2024-01", "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4-5": { "id": "anthropic/claude-opus-4-5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4-5": { "id": "anthropic/claude-sonnet-4-5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4-1": { "id": "anthropic/claude-opus-4-1", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-haiku-4-5": { "id": "anthropic/claude-haiku-4-5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-01", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 62000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-opus-4-6": { "id": "anthropic/claude-opus-4-6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "anthropic/claude-sonnet-4-6": { "id": "anthropic/claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } } } }, "qiniu-ai": { "id": "qiniu-ai", "env": ["QINIU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.qnaigc.com/v1", "name": "Qiniu", "doc": "https://developer.qiniu.com/aitokenapi", "models": { "deepseek-r1-0528": { "id": "deepseek-r1-0528", "name": "DeepSeek-R1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "doubao-1.5-thinking-pro": { "id": "doubao-1.5-thinking-pro", "name": "Doubao 1.5 Thinking Pro", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16000 } }, "qwen3-vl-30b-a3b-thinking": { "id": "qwen3-vl-30b-a3b-thinking", "name": "Qwen3-Vl 30b A3b Thinking", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-09", "last_updated": "2026-02-09", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "claude-3.5-haiku": { "id": "claude-3.5-haiku", "name": "Claude 3.5 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 } }, "deepseek-v3-0324": { "id": "deepseek-v3-0324", "name": "DeepSeek-V3-0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16000 } }, "qwen3-235b-a22b-instruct-2507": { "id": "qwen3-235b-a22b-instruct-2507", "name": "Qwen3 235b A22B Instruct 2507", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 64000 } }, "deepseek-v3": { "id": "deepseek-v3", "name": "DeepSeek-V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-13", "last_updated": "2025-08-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16000 } }, "kimi-k2": { "id": "kimi-k2", "name": "Kimi K2", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3 32B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 40000, "output": 4096 } }, "qwen3-max-preview": { "id": "qwen3-max-preview", "name": "Qwen3 Max Preview", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-06", "last_updated": "2025-09-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 } }, "claude-3.5-sonnet": { "id": "claude-3.5-sonnet", "name": "Claude 3.5 Sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-09", "last_updated": "2025-09-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8200 } }, "qwen3-next-80b-a3b-instruct": { "id": "qwen3-next-80b-a3b-instruct", "name": "Qwen3 Next 80B A3B Instruct", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-12", "last_updated": "2025-09-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 } }, "claude-4.5-haiku": { "id": "claude-4.5-haiku", "name": "Claude 4.5 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-16", "last_updated": "2025-10-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 } }, "kling-v2-6": { "id": "kling-v2-6", "name": "Kling-V2 6", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01-13", "last_updated": "2026-01-13", "modalities": { "input": ["text", "image", "video"], "output": ["video"] }, "open_weights": false, "limit": { "context": 99999999, "output": 99999999 } }, "glm-4.5": { "id": "glm-4.5", "name": "GLM 4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 98304 } }, "claude-4.1-opus": { "id": "claude-4.1-opus", "name": "Claude 4.1 Opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-06", "last_updated": "2025-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 64000 } }, "qwen3-max": { "id": "qwen3-max", "name": "Qwen3 Max", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 } }, "doubao-seed-2.0-pro": { "id": "doubao-seed-2.0-pro", "name": "Doubao Seed 2.0 Pro", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 } }, "doubao-seed-1.6": { "id": "doubao-seed-1.6", "name": "Doubao-Seed 1.6", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-15", "last_updated": "2025-08-15", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 } }, "doubao-seed-1.6-thinking": { "id": "doubao-seed-1.6-thinking", "name": "Doubao-Seed 1.6 Thinking", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-15", "last_updated": "2025-08-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 } }, "gemini-2.0-flash": { "id": "gemini-2.0-flash", "name": "Gemini 2.0 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 8192 } }, "qwen-max-2025-01-25": { "id": "qwen-max-2025-01-25", "name": "Qwen2.5-Max-2025-01-25", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 } }, "claude-4.0-sonnet": { "id": "claude-4.0-sonnet", "name": "Claude 4.0 Sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 } }, "doubao-1.5-pro-32k": { "id": "doubao-1.5-pro-32k", "name": "Doubao 1.5 Pro 32k", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 12000 } }, "qwen3-30b-a3b-instruct-2507": { "id": "qwen3-30b-a3b-instruct-2507", "name": "Qwen3 30b A3b Instruct 2507", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-04", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "qwen3-next-80b-a3b-thinking": { "id": "qwen3-next-80b-a3b-thinking", "name": "Qwen3 Next 80B A3B Thinking", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-12", "last_updated": "2025-09-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 } }, "qwen3-235b-a22b-thinking-2507": { "id": "qwen3-235b-a22b-thinking-2507", "name": "Qwen3 235B A22B Thinking 2507", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 4096 } }, "deepseek-r1": { "id": "deepseek-r1", "name": "DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "doubao-1.5-vision-pro": { "id": "doubao-1.5-vision-pro", "name": "Doubao 1.5 Vision Pro", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16000 } }, "gemini-3.0-pro-image-preview": { "id": "gemini-3.0-pro-image-preview", "name": "Gemini 3.0 Pro Image Preview", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 32768, "output": 8192 } }, "gemini-2.5-flash-image": { "id": "gemini-2.5-flash-image", "name": "Gemini 2.5 Flash Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-10-22", "last_updated": "2025-10-22", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 32768, "output": 8192 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 64000 } }, "claude-3.7-sonnet": { "id": "claude-3.7-sonnet", "name": "Claude 3.7 Sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 } }, "qwen3-30b-a3b-thinking-2507": { "id": "qwen3-30b-a3b-thinking-2507", "name": "Qwen3 30b A3b Thinking 2507", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-04", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 126000, "output": 32000 } }, "qwen2.5-vl-72b-instruct": { "id": "qwen2.5-vl-72b-instruct", "name": "Qwen 2.5 VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-06", "last_updated": "2025-08-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 } }, "doubao-seed-1.6-flash": { "id": "doubao-seed-1.6-flash", "name": "Doubao-Seed 1.6 Flash", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-15", "last_updated": "2025-08-15", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 } }, "deepseek-v3.1": { "id": "deepseek-v3.1", "name": "DeepSeek-V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "qwen3-235b-a22b": { "id": "qwen3-235b-a22b", "name": "Qwen 3 235B A22B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "qwen3-coder-480b-a35b-instruct": { "id": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-14", "last_updated": "2025-08-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 4096 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen3.5 397B A17B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-22", "last_updated": "2026-02-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 } }, "mimo-v2-flash": { "id": "mimo-v2-flash", "name": "Mimo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12-01", "release_date": "2025-12-16", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01 } }, "qwen-vl-max-2025-01-25": { "id": "qwen-vl-max-2025-01-25", "name": "Qwen VL-MAX-2025-01-25", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 } }, "qwen2.5-vl-7b-instruct": { "id": "qwen2.5-vl-7b-instruct", "name": "Qwen 2.5 VL 7B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "GLM 4.5 Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 4096 } }, "claude-4.5-opus": { "id": "claude-4.5-opus", "name": "Claude 4.5 Opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 200000 } }, "gemini-3.0-pro-preview": { "id": "gemini-3.0-pro-preview", "name": "Gemini 3.0 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image", "video", "pdf", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 } }, "doubao-seed-2.0-mini": { "id": "doubao-seed-2.0-mini", "name": "Doubao Seed 2.0 Mini", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 } }, "claude-4.0-opus": { "id": "claude-4.0-opus", "name": "Claude 4.0 Opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "gpt-oss-20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-06", "last_updated": "2025-08-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 } }, "gemini-3.0-flash-preview": { "id": "gemini-3.0-flash-preview", "name": "Gemini 3.0 Flash Preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-18", "last_updated": "2025-12-18", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 } }, "MiniMax-M1": { "id": "MiniMax-M1", "name": "MiniMax M1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 80000 } }, "qwen-turbo": { "id": "qwen-turbo", "name": "Qwen-Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 4096 } }, "qwen3-30b-a3b": { "id": "qwen3-30b-a3b", "name": "Qwen3 30B A3B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 40000, "output": 4096 } }, "claude-4.5-sonnet": { "id": "claude-4.5-sonnet", "name": "Claude 4.5 Sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 } }, "doubao-seed-2.0-lite": { "id": "doubao-seed-2.0-lite", "name": "Doubao Seed 2.0 Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 } }, "gemini-2.0-flash-lite": { "id": "gemini-2.0-flash-lite", "name": "Gemini 2.0 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 8192 } }, "doubao-seed-2.0-code": { "id": "doubao-seed-2.0-code", "name": "Doubao Seed 2.0 Code", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-11-07", "last_updated": "2025-11-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 100000 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Moonshotai/Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-01-28", "last_updated": "2026-01-28", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 } }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 100000 } }, "stepfun-ai/gelab-zero-4b-preview": { "id": "stepfun-ai/gelab-zero-4b-preview", "name": "Stepfun-Ai/Gelab Zero 4b Preview", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 } }, "x-ai/grok-code-fast-1": { "id": "x-ai/grok-code-fast-1", "name": "x-AI/Grok-Code-Fast 1", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-02", "last_updated": "2025-09-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 10000 } }, "x-ai/grok-4.1-fast-reasoning": { "id": "x-ai/grok-4.1-fast-reasoning", "name": "X-Ai/Grok 4.1 Fast Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-19", "last_updated": "2025-12-19", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 20000000, "output": 2000000 } }, "x-ai/grok-4.1-fast-non-reasoning": { "id": "x-ai/grok-4.1-fast-non-reasoning", "name": "X-Ai/Grok 4.1 Fast Non Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-19", "last_updated": "2025-12-19", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 } }, "x-ai/grok-4-fast-reasoning": { "id": "x-ai/grok-4-fast-reasoning", "name": "X-Ai/Grok-4-Fast-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-18", "last_updated": "2025-12-18", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 } }, "x-ai/grok-4-fast": { "id": "x-ai/grok-4-fast", "name": "x-AI/Grok-4-Fast", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-20", "last_updated": "2025-09-20", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 } }, "x-ai/grok-4-fast-non-reasoning": { "id": "x-ai/grok-4-fast-non-reasoning", "name": "X-Ai/Grok-4-Fast-Non-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-18", "last_updated": "2025-12-18", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 } }, "x-ai/grok-4.1-fast": { "id": "x-ai/grok-4.1-fast", "name": "x-AI/Grok-4.1-Fast", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 } }, "z-ai/autoglm-phone-9b": { "id": "z-ai/autoglm-phone-9b", "name": "Z-Ai/Autoglm Phone 9b", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 12800, "output": 4096 } }, "z-ai/glm-4.7": { "id": "z-ai/glm-4.7", "name": "Z-Ai/GLM 4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 200000 } }, "z-ai/glm-4.6": { "id": "z-ai/glm-4.6", "name": "Z-AI/GLM 4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-11", "last_updated": "2025-10-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 200000 } }, "z-ai/glm-5": { "id": "z-ai/glm-5", "name": "Z-Ai/GLM 5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "OpenAI/GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "OpenAI/GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 } }, "xiaomi/mimo-v2-flash": { "id": "xiaomi/mimo-v2-flash", "name": "Xiaomi/Mimo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12-01", "release_date": "2025-12-16", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01 } }, "stepfun/step-3.5-flash": { "id": "stepfun/step-3.5-flash", "name": "Stepfun/Step-3.5 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-02-02", "last_updated": "2026-02-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "output": 4096 } }, "meituan/longcat-flash-lite": { "id": "meituan/longcat-flash-lite", "name": "Meituan/Longcat-Flash-Lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-06", "last_updated": "2026-02-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 320000 } }, "meituan/longcat-flash-chat": { "id": "meituan/longcat-flash-chat", "name": "Meituan/Longcat-Flash-Chat", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-11-05", "last_updated": "2025-11-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 } }, "deepseek/deepseek-v3.1-terminus": { "id": "deepseek/deepseek-v3.1-terminus", "name": "DeepSeek/DeepSeek-V3.1-Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "deepseek/deepseek-v3.1-terminus-thinking": { "id": "deepseek/deepseek-v3.1-terminus-thinking", "name": "DeepSeek/DeepSeek-V3.1-Terminus-Thinking", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "deepseek/deepseek-v3.2-exp-thinking": { "id": "deepseek/deepseek-v3.2-exp-thinking", "name": "DeepSeek/DeepSeek-V3.2-Exp-Thinking", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "deepseek/deepseek-v3.2-exp": { "id": "deepseek/deepseek-v3.2-exp", "name": "DeepSeek/DeepSeek-V3.2-Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "deepseek/deepseek-v3.2-251201": { "id": "deepseek/deepseek-v3.2-251201", "name": "Deepseek/DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 } }, "deepseek/deepseek-math-v2": { "id": "deepseek/deepseek-math-v2", "name": "Deepseek/Deepseek-Math-V2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-12-04", "last_updated": "2025-12-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 160000, "output": 160000 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "Minimax/Minimax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 128000 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "Minimax/Minimax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 128000 } }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "Minimax/Minimax-M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 } }, "minimax/minimax-m2.5-highspeed": { "id": "minimax/minimax-m2.5-highspeed", "name": "Minimax/Minimax-M2.5 Highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 128000 } } } }, "alibaba-cn": { "id": "alibaba-cn", "env": ["DASHSCOPE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://dashscope.aliyuncs.com/compatible-mode/v1", "name": "Alibaba (China)", "doc": "https://www.alibabacloud.com/help/en/model-studio/models", "models": { "qwen2-5-math-72b-instruct": { "id": "qwen2-5-math-72b-instruct", "name": "Qwen2.5-Math 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 3072 }, "cost": { "input": 0.574, "output": 1.721 } }, "deepseek-r1-0528": { "id": "deepseek-r1-0528", "name": "DeepSeek R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.574, "output": 2.294 } }, "qwen3-omni-flash": { "id": "qwen3-omni-flash", "name": "Qwen3-Omni Flash", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.058, "output": 0.23, "input_audio": 3.584, "output_audio": 7.168 } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "qwen-plus": { "id": "qwen-plus", "name": "Qwen Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.115, "output": 0.287, "reasoning": 1.147 } }, "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.216, "output": 0.861 } }, "qwen2-5-coder-7b-instruct": { "id": "qwen2-5-coder-7b-instruct", "name": "Qwen2.5-Coder 7B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-11", "last_updated": "2024-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.144, "output": 0.287 } }, "deepseek-v3": { "id": "deepseek-v3", "name": "DeepSeek V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "output": 8192 }, "cost": { "input": 0.287, "output": 1.147 } }, "qwen3-omni-flash-realtime": { "id": "qwen3-omni-flash-realtime", "name": "Qwen3-Omni Flash Realtime", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.23, "output": 0.918, "input_audio": 3.584, "output_audio": 7.168 } }, "deepseek-r1-distill-llama-70b": { "id": "deepseek-r1-distill-llama-70b", "name": "DeepSeek R1 Distill Llama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.287, "output": 0.861 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 38912 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.287, "output": 1.147, "reasoning": 2.868 } }, "qwen-omni-turbo-realtime": { "id": "qwen-omni-turbo-realtime", "name": "Qwen-Omni Turbo Realtime", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-05-08", "last_updated": "2025-05-08", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0.23, "output": 0.918, "input_audio": 3.584, "output_audio": 7.168 } }, "qwen2-5-math-7b-instruct": { "id": "qwen2-5-math-7b-instruct", "name": "Qwen2.5-Math 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 3072 }, "cost": { "input": 0.144, "output": 0.287 } }, "qwen3-next-80b-a3b-instruct": { "id": "qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.144, "output": 0.574 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 128000 } } ] } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.5, "cache_write": 3.125 } }, "qwen-long": { "id": "qwen-long", "name": "Qwen Long", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01-25", "last_updated": "2025-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 10000000, "output": 8192 }, "cost": { "input": 0.072, "output": 0.287 } }, "qwen-math-turbo": { "id": "qwen-math-turbo", "name": "Qwen Math Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "output": 3072 }, "cost": { "input": 0.287, "output": 0.861 } }, "qwen3-max": { "id": "qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.861, "output": 3.441 } }, "qwen2-5-omni-7b": { "id": "qwen2-5-omni-7b", "name": "Qwen2.5-Omni 7B", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-12", "last_updated": "2024-12", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": true, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0.087, "output": 0.345, "input_audio": 5.448 } }, "qwen3-8b": { "id": "qwen3-8b", "name": "Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 38912 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.072, "output": 0.287, "reasoning": 0.717 } }, "qwen2-5-14b-instruct": { "id": "qwen2-5-14b-instruct", "name": "Qwen2.5 14B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.144, "output": 0.431 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 131072 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-14", "last_updated": "2026-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 128000 }, "cost": { "input": 0.87, "output": 3.48, "cache_read": 0.17 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "qwen3-next-80b-a3b-thinking": { "id": "qwen3-next-80b-a3b-thinking", "name": "Qwen3-Next 80B-A3B (Thinking)", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.144, "output": 1.434 } }, "qvq-max": { "id": "qvq-max", "name": "QVQ Max", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "qvq", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 1.147, "output": 4.588 } }, "qwen-plus-character": { "id": "qwen-plus-character", "name": "Qwen Plus Character", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01", "last_updated": "2024-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.115, "output": 0.287 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Moonshot Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.574, "output": 2.294 } }, "deepseek-r1": { "id": "deepseek-r1", "name": "DeepSeek R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.574, "output": 2.294 } }, "qwen3.5-flash": { "id": "qwen3.5-flash", "name": "Qwen3.5 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.172, "output": 1.72, "reasoning": 1.72 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.1, "output": 3.851, "cache_read": 0.275, "cache_write": 0 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 131072 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1875, "output": 1.125, "cache_write": 0.234375 } }, "qwen2-5-vl-72b-instruct": { "id": "qwen2-5-vl-72b-instruct", "name": "Qwen2.5-VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 2.294, "output": 6.881 } }, "qwen3-vl-plus": { "id": "qwen3-vl-plus", "name": "Qwen3-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.143353, "output": 1.433525, "reasoning": 4.300576 } }, "qwen-vl-ocr": { "id": "qwen-vl-ocr", "name": "Qwen-VL OCR", "description": "OCR model for extracting structured text from documents and screenshots", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2024-10-28", "last_updated": "2025-04-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 34096, "output": 4096 }, "cost": { "input": 0.717, "output": 0.717 } }, "qwen-mt-turbo": { "id": "qwen-mt-turbo", "name": "Qwen-MT Turbo", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 8192 }, "cost": { "input": 0.101, "output": 0.28 } }, "qwen-math-plus": { "id": "qwen-math-plus", "name": "Qwen Math Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-08-16", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "output": 3072 }, "cost": { "input": 0.574, "output": 1.721 } }, "qwen-mt-plus": { "id": "qwen-mt-plus", "name": "Qwen-MT Plus", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 8192 }, "cost": { "input": 0.259, "output": 0.775 } }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.573, "output": 3.44, "reasoning": 3.44 } }, "qwen-omni-turbo": { "id": "qwen-omni-turbo", "name": "Qwen-Omni Turbo", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01-19", "last_updated": "2025-03-26", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0.058, "output": 0.23, "input_audio": 3.584, "output_audio": 7.168 } }, "qwen2-5-72b-instruct": { "id": "qwen2-5-72b-instruct", "name": "Qwen2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.574, "output": 1.721 } }, "deepseek-r1-distill-qwen-7b": { "id": "deepseek-r1-distill-qwen-7b", "name": "DeepSeek R1 Distill Qwen 7B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.072, "output": 0.144 } }, "deepseek-v3-1": { "id": "deepseek-v3-1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.574, "output": 1.721 } }, "qwen2-5-coder-32b-instruct": { "id": "qwen2-5-coder-32b-instruct", "name": "Qwen2.5-Coder 32B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-11", "last_updated": "2024-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.287, "output": 0.861 } }, "qwen-flash": { "id": "qwen-flash", "name": "Qwen Flash", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.022, "output": 0.216 } }, "deepseek-v3-2-exp": { "id": "deepseek-v3-2-exp", "name": "DeepSeek V3.2 Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.287, "output": 0.431 } }, "moonshot-kimi-k2-instruct": { "id": "moonshot-kimi-k2-instruct", "name": "Moonshot Kimi K2 Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.574, "output": 2.294 } }, "deepseek-r1-distill-qwen-14b": { "id": "deepseek-r1-distill-qwen-14b", "name": "DeepSeek R1 Distill Qwen 14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.144, "output": 0.431 } }, "qwen3-vl-235b-a22b": { "id": "qwen3-vl-235b-a22b", "name": "Qwen3-VL 235B-A22B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.286705, "output": 1.14682, "reasoning": 2.867051 } }, "qwen-deep-research": { "id": "qwen-deep-research", "name": "Qwen Deep Research", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01", "last_updated": "2024-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 7.742, "output": 23.367 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Moonshot Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.574, "output": 2.411 } }, "qwen3-vl-30b-a3b": { "id": "qwen3-vl-30b-a3b", "name": "Qwen3-VL 30B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.108, "output": 0.431, "reasoning": 1.076 } }, "qwen-vl-max": { "id": "qwen-vl-max", "name": "Qwen-VL Max", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-08", "last_updated": "2025-08-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.23, "output": 0.574 } }, "deepseek-r1-distill-qwen-1-5b": { "id": "deepseek-r1-distill-qwen-1-5b", "name": "DeepSeek R1 Distill Qwen 1.5B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "qwen-max": { "id": "qwen-max", "name": "Qwen Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-03", "last_updated": "2025-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.345, "output": 1.377 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "qwen3-235b-a22b": { "id": "qwen3-235b-a22b", "name": "Qwen3 235B-A22B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 38912 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.287, "output": 1.147, "reasoning": 2.868 } }, "deepseek-r1-distill-llama-8b": { "id": "deepseek-r1-distill-llama-8b", "name": "DeepSeek R1 Distill Llama 8B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Moonshot Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.929, "output": 3.858 } }, "qwen3-coder-480b-a35b-instruct": { "id": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3-Coder 480B-A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.861, "output": 3.441 } }, "qwq-plus": { "id": "qwq-plus", "name": "QwQ Plus", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-03-05", "last_updated": "2025-03-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.23, "output": 0.574 } }, "qwen2-5-32b-instruct": { "id": "qwen2-5-32b-instruct", "name": "Qwen2.5 32B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.287, "output": 0.861 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.43, "output": 2.58, "reasoning": 2.58 } }, "tongyi-intent-detect-v3": { "id": "tongyi-intent-detect-v3", "name": "Tongyi Intent Detect V3", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "yi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01", "last_updated": "2024-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1024 }, "cost": { "input": 0.058, "output": 0.144 } }, "qwen3-coder-flash": { "id": "qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.144, "output": 0.574 } }, "deepseek-r1-distill-qwen-32b": { "id": "deepseek-r1-distill-qwen-32b", "name": "DeepSeek R1 Distill Qwen 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.287, "output": 0.861 } }, "qwq-32b": { "id": "qwq-32b", "name": "QwQ 32B", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-12", "last_updated": "2024-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.287, "output": 0.861 } }, "qwen3-14b": { "id": "qwen3-14b", "name": "Qwen3 14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 38912 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.144, "output": 0.574, "reasoning": 1.434 } }, "qwen3-asr-flash": { "id": "qwen3-asr-flash", "name": "Qwen3-ASR Flash", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-04", "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 53248, "output": 4096 }, "cost": { "input": 0.032, "output": 0.032 } }, "qwen-doc-turbo": { "id": "qwen-doc-turbo", "name": "Qwen Doc Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01", "last_updated": "2024-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.087, "output": 0.144 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0.86, "output": 3.15 } }, "qwen-turbo": { "id": "qwen-turbo", "name": "Qwen Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 38912 } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-11-01", "last_updated": "2025-07-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.044, "output": 0.087, "reasoning": 0.431 } }, "qwen2-5-7b-instruct": { "id": "qwen2-5-7b-instruct", "name": "Qwen2.5 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.072, "output": 0.144 } }, "qwen2-5-vl-7b-instruct": { "id": "qwen2-5-vl-7b-instruct", "name": "Qwen2.5-VL 7B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.287, "output": 0.717 } }, "qwen3.6-max-preview": { "id": "qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 131072 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 245800, "output": 65536 }, "cost": { "input": 1.32, "output": 7.9, "cache_read": 0.132 } }, "qwen-vl-plus": { "id": "qwen-vl-plus", "name": "Qwen-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-08-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.115, "output": 0.287 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } }, "MiniMax/MiniMax-M2.7": { "id": "MiniMax/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } }, "kimi/kimi-k2.5": { "id": "kimi/kimi-k2.5", "name": "kimi/kimi-k2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "siliconflow/deepseek-r1-0528": { "id": "siliconflow/deepseek-r1-0528", "name": "siliconflow/deepseek-r1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-05-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.5, "output": 2.18 } }, "siliconflow/deepseek-v3.1-terminus": { "id": "siliconflow/deepseek-v3.1-terminus", "name": "siliconflow/deepseek-v3.1-terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.27, "output": 1 } }, "siliconflow/deepseek-v3-0324": { "id": "siliconflow/deepseek-v3-0324", "name": "siliconflow/deepseek-v3-0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-26", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.25, "output": 1 } }, "siliconflow/deepseek-v3.2": { "id": "siliconflow/deepseek-v3.2", "name": "siliconflow/deepseek-v3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.27, "output": 0.42 } }, "qwen3-coder-plus": { "id": "qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Hosted Qwen coder for software agents, repo edits, and long-context code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1, "output": 5 } } } }, "regolo-ai": { "id": "regolo-ai", "env": ["REGOLO_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.regolo.ai/v1", "name": "Regolo AI", "doc": "https://docs.regolo.ai/", "models": { "llama-3.1-8b-instruct": { "id": "llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-07", "last_updated": "2025-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 120000, "output": 120000 }, "cost": { "input": 0.05, "output": 0.25 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax 2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 190000, "output": 64000 }, "cost": { "input": 0.8, "output": 3.5 } }, "mistral-small3.2": { "id": "mistral-small3.2", "name": "Mistral Small 3.2", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 120000, "output": 120000 }, "cost": { "input": 0.5, "output": 2.2 } }, "qwen3-reranker-4b": { "id": "qwen3-reranker-4b", "name": "Qwen3-Reranker-4B", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0.12, "output": 0.12 } }, "qwen3-embedding-8b": { "id": "qwen3-embedding-8b", "name": "Qwen3-Embedding-8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0.1, "output": 0.1 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.6, "output": 2.7 } }, "qwen-image": { "id": "qwen-image", "name": "Qwen-Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0.5, "output": 2 } }, "qwen3.5-122b": { "id": "qwen3.5-122b", "name": "Qwen3.5-122B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.9, "output": 3.6 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT-OSS-120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1, "output": 4.2 } }, "qwen3-coder-next": { "id": "qwen3-coder-next", "name": "Qwen3-Coder-Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.3, "output": 1.2 } }, "qwen3.5-9b": { "id": "qwen3.5-9b", "name": "Qwen3.5-9B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.15, "output": 0.6 } }, "mistral-small-4-119b": { "id": "mistral-small-4-119b", "name": "Mistral Small 4 119B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.75, "output": 3 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "GPT-OSS-20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.4, "output": 1.8 } } } }, "stackit": { "id": "stackit", "env": ["STACKIT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.openai-compat.model-serving.eu01.onstackit.cloud/v1", "name": "STACKIT", "doc": "https://docs.stackit.cloud/products/data-and-ai/ai-model-serving/basics/available-shared-models", "models": { "cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic": { "id": "cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic", "name": "Llama 3.3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.53, "output": 0.76 } }, "google/gemma-3-27b-it": { "id": "google/gemma-3-27b-it", "name": "Gemma 3 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-05-17", "last_updated": "2025-05-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 37000, "output": 4096 }, "cost": { "input": 0.53, "output": 0.76 } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.53, "output": 0.76 } }, "Qwen/Qwen3-VL-Embedding-8B": { "id": "Qwen/Qwen3-VL-Embedding-8B", "name": "Qwen3-VL Embedding 8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.09, "output": 0.09 } }, "Qwen/Qwen3-VL-235B-A22B-Instruct-FP8": { "id": "Qwen/Qwen3-VL-235B-A22B-Instruct-FP8", "name": "Qwen3-VL 235B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 218000, "output": 16384 }, "cost": { "input": 1.76, "output": 2.05 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 8192 }, "cost": { "input": 0.53, "output": 0.76 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.18, "output": 0.29 } }, "intfloat/e5-mistral-7b-instruct": { "id": "intfloat/e5-mistral-7b-instruct", "name": "E5 Mistral 7B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2023-12-11", "last_updated": "2023-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 0.02, "output": 0.02 } } } }, "vercel": { "id": "vercel", "env": ["AI_GATEWAY_API_KEY"], "npm": "@ai-sdk/gateway", "name": "Vercel AI Gateway", "doc": "https://github.com/vercel/ai/tree/5eb85cc45a259553501f535b8ac79a77d0e79223/packages/gateway", "models": { "xai/grok-imagine-video-1.5": { "id": "xai/grok-imagine-video-1.5", "name": "Grok Imagine Video 1.5", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-06-22", "last_updated": "2026-06-22", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "xai/grok-4.1-fast-reasoning": { "id": "xai/grok-4.1-fast-reasoning", "name": "Grok 4.1 Fast Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-19", "last_updated": "2025-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "xai/grok-4.20-non-reasoning-beta": { "id": "xai/grok-4.20-non-reasoning-beta", "name": "Grok 4.20 Beta Non-Reasoning", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.4 } }, "xai/grok-4.3": { "id": "xai/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "xai/grok-tts": { "id": "xai/grok-tts", "name": "Grok TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "grok", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "xai/grok-4.1-fast-non-reasoning": { "id": "xai/grok-4.1-fast-non-reasoning", "name": "Grok 4.1 Fast Non-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-19", "last_updated": "2025-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "xai/grok-voice-think-fast-1.0": { "id": "xai/grok-voice-think-fast-1.0", "name": "Grok Voice Think Fast 1.0", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "grok", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "xai/grok-imagine-video": { "id": "xai/grok-imagine-video", "name": "Grok Imagine", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-28", "last_updated": "2026-01-28", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "xai/grok-4.20-multi-agent-beta": { "id": "xai/grok-4.20-multi-agent-beta", "name": "Grok 4.20 Multi Agent Beta", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "xai/grok-stt": { "id": "xai/grok-stt", "name": "Grok STT", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "grok", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "xai/grok-4.5": { "id": "xai/grok-4.5", "name": "Grok 4.5", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 500000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.5 } }, "xai/grok-4.20-reasoning": { "id": "xai/grok-4.20-reasoning", "name": "Grok 4.20 Reasoning", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "xai/grok-4.20-reasoning-beta": { "id": "xai/grok-4.20-reasoning-beta", "name": "Grok 4.20 Beta Reasoning", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "xai/grok-imagine-video-1.5-preview": { "id": "xai/grok-imagine-video-1.5-preview", "name": "Grok Imagine Video 1.5 Preview", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-30", "last_updated": "2026-05-30", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "xai/grok-4.20-non-reasoning": { "id": "xai/grok-4.20-non-reasoning", "name": "Grok 4.20 Non-Reasoning", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "xai/grok-4.20-multi-agent": { "id": "xai/grok-4.20-multi-agent", "name": "Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "xai/grok-imagine-image": { "id": "xai/grok-imagine-image", "name": "Grok Imagine Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-28", "last_updated": "2026-02-19", "modalities": { "input": ["text"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "xai/grok-build-0.1": { "id": "xai/grok-build-0.1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-20", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2 } }, "moonshotai/kimi-k2": { "id": "moonshotai/kimi-k2", "name": "Kimi K2 Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-11", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.57, "output": 2.3 } }, "moonshotai/kimi-k2.7-code": { "id": "moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 216144, "output": 216144 }, "cost": { "input": 0.47, "output": 2, "cache_read": 0.141 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-26", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262114, "output": 262114 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-20", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "moonshotai/kimi-k2.7-code-highspeed": { "id": "moonshotai/kimi-k2.7-code-highspeed", "name": "Kimi K2.7 Code High Speed", "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-15", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 1.9, "output": 8, "cache_read": 0.38 } }, "klingai/kling-v3.0-motion-control": { "id": "klingai/kling-v3.0-motion-control", "name": "Kling v3.0 Motion Control", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-04", "last_updated": "2026-03-04", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "klingai/kling-v2.6-i2v": { "id": "klingai/kling-v2.6-i2v", "name": "Kling v2.6 Image-to-Video", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-12-21", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "klingai/kling-v2.5-turbo-t2v": { "id": "klingai/kling-v2.5-turbo-t2v", "name": "Kling v2.5 Turbo Text-to-Video", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "klingai/kling-v3.0-i2v": { "id": "klingai/kling-v3.0-i2v", "name": "Kling v3.0 Image-to-Video", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "klingai/kling-v2.5-turbo-i2v": { "id": "klingai/kling-v2.5-turbo-i2v", "name": "Kling v2.5 Turbo Image-to-Video", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "klingai/kling-v3.0-t2v": { "id": "klingai/kling-v3.0-t2v", "name": "Kling v3.0 Text-to-Video", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "klingai/kling-v2.6-motion-control": { "id": "klingai/kling-v2.6-motion-control", "name": "Kling v2.6 Motion Control", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-18", "last_updated": "2025-12-21", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "klingai/kling-v2.6-t2v": { "id": "klingai/kling-v2.6-t2v", "name": "Kling v2.6 Text-to-Video", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-12-21", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "voyage/voyage-4-lite": { "id": "voyage/voyage-4-lite", "name": "voyage-4-lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-15", "last_updated": "2026-03-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 0 } }, "voyage/voyage-law-2": { "id": "voyage/voyage-law-2", "name": "voyage-law-2", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-15", "last_updated": "2024-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "voyage/voyage-4": { "id": "voyage/voyage-4", "name": "voyage-4", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-15", "last_updated": "2026-03-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 0 } }, "voyage/voyage-code-3": { "id": "voyage/voyage-code-3", "name": "voyage-code-3", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-12-04", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "voyage/voyage-4-large": { "id": "voyage/voyage-4-large", "name": "voyage-4-large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-15", "last_updated": "2026-03-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 0 } }, "voyage/rerank-2.5": { "id": "voyage/rerank-2.5", "name": "Voyage Rerank 2.5", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 } }, "voyage/rerank-2.5-lite": { "id": "voyage/rerank-2.5-lite", "name": "Voyage Rerank 2.5 Lite", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 } }, "voyage/voyage-code-2": { "id": "voyage/voyage-code-2", "name": "voyage-code-2", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-01-01", "last_updated": "2024-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "voyage/voyage-3.5-lite": { "id": "voyage/voyage-3.5-lite", "name": "voyage-3.5-lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "voyage/voyage-3.5": { "id": "voyage/voyage-3.5", "name": "voyage-3.5", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "voyage/voyage-3-large": { "id": "voyage/voyage-3-large", "name": "voyage-3-large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-01-07", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "voyage/voyage-finance-2": { "id": "voyage/voyage-finance-2", "name": "voyage-finance-2", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "voyage", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-06-03", "last_updated": "2024-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "mistral/mistral-nemo": { "id": "mistral/mistral-nemo", "name": "Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-07-18", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.15 } }, "mistral/codestral-embed": { "id": "mistral/codestral-embed", "name": "Codestral Embed", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "codestral-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "mistral/mistral-embed": { "id": "mistral/mistral-embed", "name": "Mistral Embed", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "mistral-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-12-11", "last_updated": "2023-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "mistral/devstral-small": { "id": "mistral/devstral-small", "name": "Devstral Small 1.1", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-05-21", "last_updated": "2025-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.1, "output": 0.3 } }, "mistral/mistral-large-3": { "id": "mistral/mistral-large-3", "name": "Mistral Large 3", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.5, "output": 1.5 } }, "mistral/mistral-medium-3.5": { "id": "mistral/mistral-medium-3.5", "name": "Mistral Medium Latest", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-05-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1.5, "output": 7.5 } }, "mistral/mistral-medium": { "id": "mistral/mistral-medium", "name": "Mistral Medium 3.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.4, "output": 2 } }, "mistral/devstral-small-2": { "id": "mistral/devstral-small-2", "name": "Devstral Small 2", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-09", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3 } }, "mistral/ministral-14b": { "id": "mistral/ministral-14b", "name": "Ministral 14B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-02", "last_updated": "2025-12-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 0.2 } }, "mistral/devstral-2": { "id": "mistral/devstral-2", "name": "Devstral 2", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.4, "output": 2 } }, "mistral/mistral-small": { "id": "mistral/mistral-small", "name": "Mistral Small (latest)", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2024-09-17", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4000 }, "cost": { "input": 0.1, "output": 0.3 } }, "mistral/ministral-8b": { "id": "mistral/ministral-8b", "name": "Ministral 8B (latest)", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.1, "output": 0.1 } }, "mistral/codestral": { "id": "mistral/codestral", "name": "Codestral (latest)", "description": "Mistral code model for completions, refactors, and developer IDE workflows", "family": "codestral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-05-29", "last_updated": "2025-01-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 4096 }, "cost": { "input": 0.3, "output": 0.9 } }, "mistral/pixtral-12b": { "id": "mistral/pixtral-12b", "name": "Pixtral 12B", "description": "Mistral vision-language model for image understanding and multimodal chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-09-01", "last_updated": "2024-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.15 } }, "mistral/pixtral-large": { "id": "mistral/pixtral-large", "name": "Pixtral Large (latest)", "description": "Mistral's larger vision model for document-heavy image understanding and chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2024-11-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 2, "output": 6 } }, "mistral/ministral-3b": { "id": "mistral/ministral-3b", "name": "Ministral 3B (latest)", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.04, "output": 0.04 } }, "mistral/magistral-small": { "id": "mistral/magistral-small", "name": "Magistral Small", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-small", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-03-17", "last_updated": "2025-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.5, "output": 1.5 } }, "mistral/magistral-medium": { "id": "mistral/magistral-medium", "name": "Magistral Medium (latest)", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-medium", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-03-17", "last_updated": "2025-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2, "output": 5 } }, "google/gemini-embedding-2": { "id": "google/gemini-embedding-2", "name": "Gemini Embedding 2", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "gemini-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "google/text-multilingual-embedding-002": { "id": "google/text-multilingual-embedding-002", "name": "Text Multilingual Embedding 002", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-03-01", "last_updated": "2024-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.03 } }, "google/gemini-3.1-flash-image": { "id": "google/gemini-3.1-flash-image", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05 } }, "google/gemini-3-flash": { "id": "google/gemini-3-flash", "name": "Gemini 3 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05 } }, "google/veo-3.1-generate-001": { "id": "google/veo-3.1-generate-001", "name": "Veo 3.1", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-15", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.14, "output": 0.4 } }, "google/veo-3.0-generate-001": { "id": "google/veo-3.0-generate-001", "name": "Veo 3.0", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-20", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "google/veo-3.0-fast-generate-001": { "id": "google/veo-3.0-fast-generate-001", "name": "Veo 3.0 Fast Generate", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-07-31", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "google/text-embedding-005": { "id": "google/text-embedding-005", "name": "Text Embedding 005", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-01", "last_updated": "2024-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "google/gemini-embedding-001": { "id": "google/gemini-embedding-001", "name": "Gemini Embedding 001", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "gemini-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "google/gemini-3.1-flash-lite-image": { "id": "google/gemini-3.1-flash-lite-image", "name": "Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite)", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 4096 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.03 } }, "google/gemini-2.5-flash-image": { "id": "google/gemini-2.5-flash-image", "name": "Nano Banana (Gemini 2.5 Flash Image)", "description": "Nano Banana image model for fast generation, edits, and character-consistent assets", "family": "gemini-flash", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 32768, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03 } }, "google/imagen-4.0-fast-generate-001": { "id": "google/imagen-4.0-fast-generate-001", "name": "Imagen 4 Fast", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-06-01", "last_updated": "2025-06", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01 } }, "google/gemini-omni-flash-preview": { "id": "google/gemini-omni-flash-preview", "name": "Gemini Omni Flash Preview", "description": "Omni-modal model for text, vision, audio, and multimodal agent tasks", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 57920 }, "cost": { "input": 1.5, "output": 9 } }, "google/gemini-3.1-flash-image-preview": { "id": "google/gemini-3.1-flash-image-preview", "name": "Gemini 3.1 Flash Image Preview (Nano Banana 2)", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05 } }, "google/imagen-4.0-ultra-generate-001": { "id": "google/imagen-4.0-ultra-generate-001", "name": "Imagen 4 Ultra", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-24", "last_updated": "2025-05-24", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015 } }, "google/gemini-3-pro-preview": { "id": "google/gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/veo-3.1-fast-generate-001": { "id": "google/veo-3.1-fast-generate-001", "name": "Veo 3.1 Fast Generate", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-15", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "google/gemini-3-pro-image": { "id": "google/gemini-3-pro-image", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 32768 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.03 } }, "google/imagen-4.0-generate-001": { "id": "google/imagen-4.0-generate-001", "name": "Imagen 4", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-20", "last_updated": "2025-05-22", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "prodia/flux-fast-schnell": { "id": "prodia/flux-fast-schnell", "name": "Flux Schnell", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-02", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "openai/gpt-oss-safeguard-20b": { "id": "openai/gpt-oss-safeguard-20b", "name": "gpt-oss-safeguard-20b", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-10-29", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 65536, "output": 65536 }, "cost": { "input": 0.075, "output": 0.3, "cache_read": 0.037 } }, "openai/gpt-3.5-turbo-instruct": { "id": "openai/gpt-3.5-turbo-instruct", "name": "GPT-3.5 Turbo Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-09", "release_date": "2023-09-18", "last_updated": "2023-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 4096, "output": 4096 }, "cost": { "input": 1.5, "output": 2 } }, "openai/gpt-5.2-chat": { "id": "openai/gpt-5.2-chat", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-11", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 111616, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/text-embedding-3-large": { "id": "openai/text-embedding-3-large", "name": "text-embedding-3-large", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 6656, "output": 1536 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "GPT 5.2 ", "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "openai/gpt-4o-mini-search-preview": { "id": "openai/gpt-4o-mini-search-preview", "name": "GPT 4o Mini Search Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-09", "release_date": "2025-03-12", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 111616, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "openai/gpt-5-chat": { "id": "openai/gpt-5-chat", "name": "GPT-5 Chat", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 128000, "input": 111616, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-3.5-turbo": { "id": "openai/gpt-3.5-turbo", "name": "GPT-3.5 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2021-09", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "input": 12289, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5 } }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", "name": "GPT-5 pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 128000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "openai/o3-pro": { "id": "openai/o3-pro", "name": "o3 Pro", "description": "High-effort o3 tier for difficult technical reasoning and careful answers", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 100000, "output": 100000 }, "cost": { "input": 20, "output": 80 } }, "openai/gpt-4o-transcribe": { "id": "openai/gpt-4o-transcribe", "name": "GPT-4o Transcribe", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-03-13", "last_updated": "2024-03-13", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 2.5, "output": 10 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT 5.4 Nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "openai/gpt-5.3-chat": { "id": "openai/gpt-5.3-chat", "name": "GPT-5.3 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 111616, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1-thinking": { "id": "openai/gpt-5.1-thinking", "name": "GPT 5.1 Thinking", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-12", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT-5.1-Codex", "description": "Codex GPT for repository edits, code review, and practical software agents", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-12", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.1-codex-max": { "id": "openai/gpt-5.1-codex-max", "name": "GPT 5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-19", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-4o-mini-transcribe": { "id": "openai/gpt-4o-mini-transcribe", "name": "GPT-4o mini Transcribe", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "o-mini", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-03-13", "last_updated": "2024-03-13", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 1.25, "output": 5 } }, "openai/text-embedding-ada-002": { "id": "openai/text-embedding-ada-002", "name": "text-embedding-ada-002", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2022-12-15", "last_updated": "2022-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 6656, "output": 1536 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT 5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/text-embedding-3-small": { "id": "openai/text-embedding-3-small", "name": "text-embedding-3-small", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 6656, "output": 1536 } }, "openai/gpt-realtime-1.5": { "id": "openai/gpt-realtime-1.5", "name": "GPT-Realtime-1.5", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 4, "output": 16, "cache_read": 0.4 } }, "openai/gpt-5.6-luna": { "id": "openai/gpt-5.6-luna", "name": "GPT 5.6 Luna", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-12", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-realtime-2.1": { "id": "openai/gpt-realtime-2.1", "name": "gpt-realtime-2.1", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"] } ], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-09-30", "release_date": "2026-07-09", "last_updated": "2026-07-06", "modalities": { "input": ["text", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 128000, "input": 96000, "output": 32000 }, "cost": { "input": 4, "output": 24, "cache_read": 0.4 } }, "openai/gpt-5.6-terra": { "id": "openai/gpt-5.6-terra", "name": "GPT 5.6 Terra", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 3.125 } }, "openai/gpt-image-1.5": { "id": "openai/gpt-image-1.5", "name": "GPT Image 1.5", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 32, "cache_read": 1.25 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.1, "output": 0.5 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT 5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT 5.4 Mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai/gpt-realtime-mini": { "id": "openai/gpt-realtime-mini", "name": "GPT-Realtime mini", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-10", "last_updated": "2025-10-10", "modalities": { "input": ["text", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06 } }, "openai/tts-1": { "id": "openai/tts-1", "name": "TTS-1", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-11-06", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "openai/whisper-1": { "id": "openai/whisper-1", "name": "Whisper", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2022-09-21", "last_updated": "2022-09-21", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "openai/tts-1-hd": { "id": "openai/tts-1-hd", "name": "TTS-1 HD", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-11-06", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "openai/o3-deep-research": { "id": "openai/o3-deep-research", "name": "o3-deep-research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-06-26", "last_updated": "2024-06-26", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 100000, "output": 100000 }, "cost": { "input": 10, "output": 40, "cache_read": 2.5 } }, "openai/gpt-image-1": { "id": "openai/gpt-image-1", "name": "GPT Image 1", "description": "OpenAI image model for production generation, edits, and brand-safe visual workflows", "family": "gpt-image", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-25", "last_updated": "2025-04-24", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 40, "cache_read": 1.25 } }, "openai/gpt-realtime-2": { "id": "openai/gpt-realtime-2", "name": "gpt-realtime-2", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 4, "output": 24, "cache_read": 0.4 } }, "openai/gpt-image-1-mini": { "id": "openai/gpt-image-1-mini", "name": "GPT Image 1 Mini", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 2, "output": 8, "cache_read": 0.2 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "GPT 5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT 5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12-01", "release_date": "2026-04-24", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 872000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "input": 122880, "output": 8192 }, "cost": { "input": 0.05, "output": 0.2 } }, "openai/gpt-5.6-sol": { "id": "openai/gpt-5.6-sol", "name": "GPT 5.6 Sol", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT-5.2-Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-18", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1-instant": { "id": "openai/gpt-5.1-instant", "name": "GPT-5.1 Instant", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-12", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 111616, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-image-2": { "id": "openai/gpt-image-2", "name": "GPT Image 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 30, "cache_read": 1.25 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT 5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12-01", "release_date": "2026-04-24", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 872000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "openai/gpt-4-turbo": { "id": "openai/gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/o1": { "id": "openai/o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/o3": { "id": "openai/o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "zai/glm-4.7": { "id": "zai/glm-4.7", "name": "GLM 4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 120000 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.12 } }, "zai/glm-4.5v": { "id": "zai/glm-4.5v", "name": "GLM 4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 66000, "output": 16000 }, "cost": { "input": 0.6, "output": 1.8, "cache_read": 0.11 } }, "zai/glm-4.5": { "id": "zai/glm-4.5", "name": "GLM 4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 96000 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "zai/glm-4.7-flashx": { "id": "zai/glm-4.7-flashx", "name": "GLM 4.7 FlashX", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.06, "output": 0.4, "cache_read": 0.01 } }, "zai/glm-5.1": { "id": "zai/glm-5.1", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202000, "output": 202000 }, "cost": { "input": 1.3, "output": 4.3, "cache_read": 0.26 } }, "zai/glm-4.6": { "id": "zai/glm-4.6", "name": "GLM 4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 96000 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "zai/glm-5.2": { "id": "zai/glm-5.2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-16", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1040000, "output": 128000 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "zai/glm-4.6v": { "id": "zai/glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-30", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 24000 }, "cost": { "input": 0.3, "output": 0.9, "cache_read": 0.05 } }, "zai/glm-5.2-fast": { "id": "zai/glm-5.2-fast", "name": "GLM 5.2 Fast", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-06-23", "last_updated": "2026-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2.1, "output": 6.6, "cache_read": 0.21 } }, "zai/glm-5v-turbo": { "id": "zai/glm-5v-turbo", "name": "GLM 5V Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "zai/glm-4.6v-flash": { "id": "zai/glm-4.6v-flash", "name": "GLM-4.6V-Flash", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 24000 } }, "zai/glm-4.5-air": { "id": "zai/glm-4.5-air", "name": "GLM 4.5 Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 96000 }, "cost": { "input": 0.2, "output": 1.1, "cache_read": 0.03 } }, "zai/glm-4.7-flash": { "id": "zai/glm-4.7-flash", "name": "GLM 4.7 Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131000 }, "cost": { "input": 0.07, "output": 0.4 } }, "zai/glm-5": { "id": "zai/glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 131100 }, "cost": { "input": 0.95, "output": 3.15, "cache_read": 0.2 } }, "zai/glm-5-turbo": { "id": "zai/glm-5-turbo", "name": "GLM 5 Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202800, "output": 131100 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "bytedance/seedream-5.0-lite": { "id": "bytedance/seedream-5.0-lite", "name": "Seedream 5.0 Lite", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-01-28", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seedance-2.0-fast": { "id": "bytedance/seedance-2.0-fast", "name": "Seedance 2.0 Fast", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "seed", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-14", "last_updated": "2026-04-14", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seedance-v1.0-pro": { "id": "bytedance/seedance-v1.0-pro", "name": "Seedance v1.0 Pro", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-06-11", "last_updated": "2025-06-11", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seed-1.6": { "id": "bytedance/seed-1.6", "name": "Seed 1.6", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-01", "last_updated": "2025-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.05 } }, "bytedance/seedance-2.0": { "id": "bytedance/seedance-2.0", "name": "Seedance 2.0", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "seed", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-14", "last_updated": "2026-04-14", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seedance-v1.5-pro": { "id": "bytedance/seedance-v1.5-pro", "name": "Seedance v1.5 Pro", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seedance-v1.0-pro-fast": { "id": "bytedance/seedance-v1.0-pro-fast", "name": "Seedance v1.0 Pro Fast", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-24", "last_updated": "2025-10-31", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seedream-4.0": { "id": "bytedance/seedream-4.0", "name": "Seedream 4.0", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-09", "last_updated": "2025-08-28", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seedream-4.5": { "id": "bytedance/seedream-4.5", "name": "Seedream 4.5", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-11-28", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seedream-5.0-pro": { "id": "bytedance/seedream-5.0-pro", "name": "Seedream 5.0 Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-07-11", "last_updated": "2026-07-11", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bytedance/seed-1.8": { "id": "bytedance/seed-1.8", "name": "Seed 1.8", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-01", "last_updated": "2025-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.05 } }, "morph/morph-v3-large": { "id": "morph/morph-v3-large", "name": "Morph v3 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-08-15", "last_updated": "2024-08-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.9, "output": 1.9 } }, "morph/morph-v3-fast": { "id": "morph/morph-v3-fast", "name": "Morph v3 Fast", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-08-15", "last_updated": "2024-08-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16000, "output": 16000 }, "cost": { "input": 0.8, "output": 1.2 } }, "sakana/fugu-ultra": { "id": "sakana/fugu-ultra", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "family": "aura", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-21", "last_updated": "2026-06-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "nvidia/nemotron-3-ultra-550b-a55b": { "id": "nvidia/nemotron-3-ultra-550b-a55b", "name": "Nemotron 3 Ultra", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.12 } }, "nvidia/nemotron-3-nano-30b-a3b": { "id": "nvidia/nemotron-3-nano-30b-a3b", "name": "Nemotron 3 Nano 30B A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.05, "output": 0.24 } }, "nvidia/nemotron-nano-9b-v2": { "id": "nvidia/nemotron-nano-9b-v2", "name": "Nvidia Nemotron Nano 9B V2", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-08-18", "last_updated": "2025-08-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.06, "output": 0.23 } }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "NVIDIA Nemotron 3 Super 120B A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0.15, "output": 0.65 } }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", "name": "Nvidia Nemotron Nano 12B V2 VL", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.2, "output": 0.6 } }, "xiaomi/mimo-v2.5": { "id": "xiaomi/mimo-v2.5", "name": "MiMo M2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo-v2.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 131100 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "xiaomi/mimo-v2-flash": { "id": "xiaomi/mimo-v2-flash", "name": "MiMo V2 Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-16", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01 } }, "xiaomi/mimo-v2-pro": { "id": "xiaomi/mimo-v2-pro", "name": "MiMo V2 Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2 } }, "xiaomi/mimo-v2.5-pro": { "id": "xiaomi/mimo-v2.5-pro", "name": "MiMo V2.5 Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo-v2.5-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 131000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036 } }, "quiverai/arrow-1.1": { "id": "quiverai/arrow-1.1", "name": "Arrow 1.1", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 } }, "inception/mercury-coder-small": { "id": "inception/mercury-coder-small", "name": "Mercury Coder Small Beta", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "mercury", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-02-26", "last_updated": "2025-02-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 16384 }, "cost": { "input": 0.25, "output": 1 } }, "inception/mercury-2": { "id": "inception/mercury-2", "name": "Mercury 2", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "mercury", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-24", "last_updated": "2026-03-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.25, "output": 0.75, "cache_read": 0.024999999999999998 } }, "anthropic/claude-3.5-haiku": { "id": "anthropic/claude-3.5-haiku", "name": "Claude 3.5 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-11-04", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-sonnet-5": { "id": "anthropic/claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-31", "release_date": "2026-06-29", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-fable-5": { "id": "anthropic/claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-31", "release_date": "2026-07-01", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "anthropic/claude-opus-4.5": { "id": "anthropic/claude-opus-4.5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "anthropic/claude-3-haiku": { "id": "anthropic/claude-3-haiku", "name": "Claude Haiku 3", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-03-13", "last_updated": "2024-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.03, "cache_write": 0.3 } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-opus-4.1": { "id": "anthropic/claude-opus-4.1", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-sonnet-4.5": { "id": "anthropic/claude-sonnet-4.5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "cohere/rerank-v3.5": { "id": "cohere/rerank-v3.5", "name": "Cohere Rerank 3.5", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-12-02", "last_updated": "2024-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "output": 4096 } }, "cohere/command-a": { "id": "cohere/command-a", "name": "Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 8000 }, "cost": { "input": 2.5, "output": 10 } }, "cohere/rerank-v4-fast": { "id": "cohere/rerank-v4-fast", "name": "Cohere Rerank 4 Fast", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 } }, "cohere/embed-v4.0": { "id": "cohere/embed-v4.0", "name": "Embed v4.0", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "cohere-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 1536 } }, "cohere/rerank-v4-pro": { "id": "cohere/rerank-v4-pro", "name": "Cohere Rerank 4 Pro", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 } }, "stepfun/step-3.7-flash": { "id": "stepfun/step-3.7-flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "family": "step", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-28", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 1.15, "cache_read": 0.04 } }, "stepfun/step-3.5-flash": { "id": "stepfun/step-3.5-flash", "name": "StepFun 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "family": "step", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262114, "output": 262114 }, "cost": { "input": 0.09, "output": 0.3, "cache_read": 0.02 } }, "interfaze/interfaze-beta": { "id": "interfaze/interfaze-beta", "name": "Interfaze Beta", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "temperature": true, "release_date": "2025-10-07", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 1.5, "output": 3.5 } }, "bfl/flux-kontext-max": { "id": "bfl/flux-kontext-max", "name": "FLUX.1 Kontext Max", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-29", "last_updated": "2025-06", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "bfl/flux-2-flex": { "id": "bfl/flux-2-flex", "name": "FLUX.2 [flex]", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-25", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bfl/flux-pro-1.1-ultra": { "id": "bfl/flux-pro-1.1-ultra", "name": "FLUX1.1 [pro] Ultra", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-11-01", "last_updated": "2024-11", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "bfl/flux-2-max": { "id": "bfl/flux-2-max", "name": "FLUX.2 [max]", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 67300, "output": 67300 } }, "bfl/flux-pro-1.1": { "id": "bfl/flux-pro-1.1", "name": "FLUX1.1 [pro]", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-02", "last_updated": "2024-10", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "bfl/flux-pro-1.0-fill": { "id": "bfl/flux-pro-1.0-fill", "name": "FLUX.1 Fill [pro]", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-01", "last_updated": "2024-10", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "bfl/flux-2-klein-4b": { "id": "bfl/flux-2-klein-4b", "name": "FLUX.2 [klein] 4B", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-15", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bfl/flux-2-klein-9b": { "id": "bfl/flux-2-klein-9b", "name": "FLUX.2 [klein] 9B", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-15", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "bfl/flux-kontext-pro": { "id": "bfl/flux-kontext-pro", "name": "FLUX.1 Kontext Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-29", "last_updated": "2025-06", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "bfl/flux-2-pro": { "id": "bfl/flux-2-pro", "name": "FLUX.2 [pro]", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-25", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 67300, "output": 67300 } }, "recraft/recraft-v4.1-pro": { "id": "recraft/recraft-v4.1-pro", "name": "Recraft V4.1 Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-14", "last_updated": "2026-05-14", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "recraft/recraft-v4.1": { "id": "recraft/recraft-v4.1", "name": "Recraft V4.1", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-14", "last_updated": "2026-05-14", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "recraft/recraft-v4": { "id": "recraft/recraft-v4", "name": "Recraft V4", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "recraft/recraft-v4-pro": { "id": "recraft/recraft-v4-pro", "name": "Recraft V4 Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "recraft/recraft-v2": { "id": "recraft/recraft-v2", "name": "Recraft V2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-03-13", "last_updated": "2024-03", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "recraft/recraft-v3": { "id": "recraft/recraft-v3", "name": "Recraft V3", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-30", "last_updated": "2024-10", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 512, "output": 0 } }, "recraft/recraft-v4.1-utility-pro": { "id": "recraft/recraft-v4.1-utility-pro", "name": "Recraft V4.1 Utility Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-14", "last_updated": "2026-05-14", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "recraft/recraft-v4.1-utility": { "id": "recraft/recraft-v4.1-utility", "name": "Recraft V4.1 Utility", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "recraft", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-14", "last_updated": "2026-05-14", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "arcee-ai/trinity-large-preview": { "id": "arcee-ai/trinity-large-preview", "name": "Trinity Large Preview", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "trinity", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2026-01-27", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.25, "output": 1 } }, "arcee-ai/trinity-large-thinking": { "id": "arcee-ai/trinity-large-thinking", "name": "Trinity Large Thinking", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "trinity", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262100, "output": 80000 }, "cost": { "input": 0.25, "output": 0.8999999999999999 } }, "arcee-ai/trinity-mini": { "id": "arcee-ai/trinity-mini", "name": "Trinity Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "trinity", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-01", "last_updated": "2025-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.045, "output": 0.15 } }, "perplexity/sonar-reasoning-pro": { "id": "perplexity/sonar-reasoning-pro", "name": "Sonar Reasoning Pro", "description": "Web-grounded reasoning model for multi-step research and cited answers", "family": "sonar-reasoning", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-09", "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "output": 8000 } }, "perplexity/sonar": { "id": "perplexity/sonar", "name": "Sonar", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-02", "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "output": 8000 } }, "perplexity/sonar-pro": { "id": "perplexity/sonar-pro", "name": "Sonar Pro", "description": "Advanced Sonar search model for deeper research and cited synthesis", "family": "sonar-pro", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8000 } }, "amazon/titan-embed-text-v2": { "id": "amazon/titan-embed-text-v2", "name": "Titan Text Embeddings V2", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "titan-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-30", "last_updated": "2024-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 } }, "amazon/nova-2-lite": { "id": "amazon/nova-2-lite", "name": "Nova 2 Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "nova", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-02", "last_updated": "2024-12-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075 } }, "amazon/nova-lite": { "id": "amazon/nova-lite", "name": "Nova Lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-lite", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 8192 }, "cost": { "input": 0.06, "output": 0.24, "cache_read": 0.015 } }, "amazon/nova-micro": { "id": "amazon/nova-micro", "name": "Nova Micro", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-micro", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.035, "output": 0.14, "cache_read": 0.00875 } }, "amazon/nova-pro": { "id": "amazon/nova-pro", "name": "Nova Pro", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova-pro", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 8192 }, "cost": { "input": 0.8, "output": 3.2, "cache_read": 0.2 } }, "alibaba/qwen3-vl-thinking": { "id": "alibaba/qwen3-vl-thinking", "name": "Qwen3 VL Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-23", "last_updated": "2025-09-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.4, "output": 4 } }, "alibaba/qwen3-coder-plus": { "id": "alibaba/qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Hosted Qwen coder for software agents, repo edits, and long-context code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1, "output": 5, "cache_read": 0.2 } }, "alibaba/wan-v2.6-r2v": { "id": "alibaba/wan-v2.6-r2v", "name": "Wan v2.6 Reference-to-Video", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/qwen3-embedding-0.6b": { "id": "alibaba/qwen3-embedding-0.6b", "name": "Qwen3 Embedding 0.6B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 } }, "alibaba/qwen3-max-preview": { "id": "alibaba/qwen3-max-preview", "name": "Qwen3 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-05", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 1.2, "output": 6, "cache_read": 0.24 } }, "alibaba/qwen3-embedding-8b": { "id": "alibaba/qwen3-embedding-8b", "name": "Qwen3 Embedding 8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 } }, "alibaba/qwen3-next-80b-a3b-instruct": { "id": "alibaba/qwen3-next-80b-a3b-instruct", "name": "Qwen3 Next 80B A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-11", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 1.2 } }, "alibaba/qwen3.7-plus": { "id": "alibaba/qwen3.7-plus", "name": "Qwen 3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen3.7-plus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 262144 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.08, "cache_write": 0.5 } }, "alibaba/wan-v2.7-r2v": { "id": "alibaba/wan-v2.7-r2v", "name": "Wan v2.7 Reference-to-Video", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/qwen3-vl-instruct": { "id": "alibaba/qwen3-vl-instruct", "name": "Qwen3 VL Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 129024 }, "cost": { "input": 0.4, "output": 1.6 } }, "alibaba/qwen3.7-max": { "id": "alibaba/qwen3.7-max", "name": "Qwen 3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 262144 } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991000, "output": 64000 }, "cost": { "input": 1.25, "output": 3.75, "cache_read": 0.25, "cache_write": 1.5625 } }, "alibaba/wan-v2.5-t2v-preview": { "id": "alibaba/wan-v2.5-t2v-preview", "name": "Wan v2.5 Text-to-Video Preview", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/qwen3-max": { "id": "alibaba/qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 1.2, "output": 6, "cache_read": 0.24 } }, "alibaba/qwen3-next-80b-a3b-thinking": { "id": "alibaba/qwen3-next-80b-a3b-thinking", "name": "Qwen3 Next 80B A3B Thinking", "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1 } ], "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-11", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 1.2 } }, "alibaba/qwen3-embedding-4b": { "id": "alibaba/qwen3-embedding-4b", "name": "Qwen3 Embedding 4B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 } }, "alibaba/qwen3.5-flash": { "id": "alibaba/qwen3.5-flash", "name": "Qwen 3.5 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.001, "cache_write": 0.125 } }, "alibaba/qwen3-coder": { "id": "alibaba/qwen3-coder", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-22", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.5, "output": 7.5, "cache_read": 0.3 } }, "alibaba/qwen-3-235b": { "id": "alibaba/qwen-3-235b", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-28", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.22, "output": 0.88 } }, "alibaba/qwen3.5-plus": { "id": "alibaba/qwen3.5-plus", "name": "Qwen 3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.4, "output": 2.4, "cache_read": 0.04, "cache_write": 0.5 } }, "alibaba/wan-v2.6-t2v": { "id": "alibaba/wan-v2.6-t2v", "name": "Wan v2.6 Text-to-Video", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/qwen3-max-thinking": { "id": "alibaba/qwen3-max-thinking", "name": "Qwen 3 Max Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-23", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 1.2, "output": 6, "cache_read": 0.24 } }, "alibaba/wan-v2.6-i2v-flash": { "id": "alibaba/wan-v2.6-i2v-flash", "name": "Wan v2.6 Image-to-Video Flash", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/wan-v2.6-r2v-flash": { "id": "alibaba/wan-v2.6-r2v-flash", "name": "Wan v2.6 Reference-to-Video Flash", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/qwen3-coder-next": { "id": "alibaba/qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-07-22", "last_updated": "2026-02-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.5, "output": 1.2 } }, "alibaba/qwen3.6-27b": { "id": "alibaba/qwen3.6-27b", "name": "Qwen 3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 131072 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.6, "output": 3.6 } }, "alibaba/wan-v2.6-i2v": { "id": "alibaba/wan-v2.6-i2v", "name": "Wan v2.6 Image-to-Video", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/wan-v2.7-t2v": { "id": "alibaba/wan-v2.7-t2v", "name": "Wan v2.7 Text-to-Video", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "alibaba/qwen-3-30b": { "id": "alibaba/qwen-3-30b", "name": "Qwen3-30B-A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-28", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 40960, "output": 16384 }, "cost": { "input": 0.12, "output": 0.5 } }, "alibaba/qwen3-235b-a22b-thinking": { "id": "alibaba/qwen3-235b-a22b-thinking", "name": "Qwen3 235B A22B Thinking 2507", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-04", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.4, "output": 4 } }, "alibaba/qwen3-vl-235b-a22b-instruct": { "id": "alibaba/qwen3-vl-235b-a22b-instruct", "name": "Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-23", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 129024 }, "cost": { "input": 0.4, "output": 1.6 } }, "alibaba/qwen-3-14b": { "id": "alibaba/qwen-3-14b", "name": "Qwen3-14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-28", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 40960, "output": 16384 }, "cost": { "input": 0.12, "output": 0.24 } }, "alibaba/qwen-3-32b": { "id": "alibaba/qwen-3-32b", "name": "Qwen 3.32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 38912 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-28", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.16, "output": 0.64 } }, "alibaba/qwen-3.6-max-preview": { "id": "alibaba/qwen-3.6-max-preview", "name": "Qwen 3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 131072 } ], "tool_call": true, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 240000, "output": 64000 }, "cost": { "input": 1.3, "output": 7.8, "cache_read": 0.26, "cache_write": 1.625 } }, "alibaba/qwen3-coder-30b-a3b": { "id": "alibaba/qwen3-coder-30b-a3b", "name": "Qwen 3 Coder 30B A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-31", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.15, "output": 0.6 } }, "alibaba/qwen3.6-plus": { "id": "alibaba/qwen3.6-plus", "name": "Qwen 3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 131072 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.1, "cache_write": 0.625 } }, "meituan/longcat-flash-thinking-2601": { "id": "meituan/longcat-flash-thinking-2601", "name": "LongCat Flash Thinking 2601", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "longcat", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2026-01-15", "last_updated": "2026-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 } }, "meituan/longcat-flash-chat": { "id": "meituan/longcat-flash-chat", "name": "LongCat Flash Chat", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "longcat", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-01", "last_updated": "2025-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 100000 } }, "meta/muse-spark-1.1": { "id": "meta/muse-spark-1.1", "name": "Muse Spark 1.1", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "muse", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 1048576 }, "cost": { "input": 1.25, "output": 4.25, "cache_read": 0.15 } }, "meta/llama-3.2-1b": { "id": "meta/llama-3.2-1b", "name": "Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.1, "output": 0.1 } }, "meta/llama-3.2-11b": { "id": "meta/llama-3.2-11b", "name": "Llama 3.2 11B Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.16, "output": 0.16 } }, "meta/llama-3.1-8b": { "id": "meta/llama-3.1-8b", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.22, "output": 0.22 } }, "meta/llama-3.2-90b": { "id": "meta/llama-3.2-90b", "name": "Llama 3.2 90B Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.72, "output": 0.72 } }, "meta/llama-3.1-70b": { "id": "meta/llama-3.1-70b", "name": "Llama 3.1 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.72, "output": 0.72 } }, "meta/llama-3.2-3b": { "id": "meta/llama-3.2-3b", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.15, "output": 0.15 } }, "meta/llama-4-scout": { "id": "meta/llama-4-scout", "name": "Llama-4-Scout-17B-16E-Instruct-FP8", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.3-70b": { "id": "meta/llama-3.3-70b", "name": "Llama-3.3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-4-maverick": { "id": "meta/llama-4-maverick", "name": "Llama-4-Maverick-17B-128E-Instruct-FP8", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-23", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "deepseek/deepseek-v3.1-terminus": { "id": "deepseek/deepseek-v3.1-terminus", "name": "DeepSeek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.27, "output": 1, "cache_read": 0.135 } }, "deepseek/deepseek-v3": { "id": "deepseek/deepseek-v3", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-12-26", "last_updated": "2024-12-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.27, "output": 1.12, "cache_read": 0.135 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-23", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036 } }, "deepseek/deepseek-v3.2-thinking": { "id": "deepseek/deepseek-v3.2-thinking", "name": "DeepSeek V3.2 Thinking", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8000 }, "cost": { "input": 0.62, "output": 1.85 } }, "deepseek/deepseek-v3.1": { "id": "deepseek/deepseek-v3.1", "name": "DeepSeek-V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.6, "output": 1.7 } }, "deepseek/deepseek-v3.2": { "id": "deepseek/deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8000 }, "cost": { "input": 0.28, "output": 0.42, "cache_read": 0.028 } }, "deepseek/deepseek-r1": { "id": "deepseek/deepseek-r1", "name": "DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 1.35, "output": 5.4 } }, "minimax/minimax-m2.7-highspeed": { "id": "minimax/minimax-m2.7-highspeed", "name": "MiniMax M2.7 High Speed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131100 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "MiniMax M2.1", "description": "Earlier MiniMax agent model for practical coding and productivity tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", "name": "MiniMax M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax-m3", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-05-31", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 205000, "output": 205000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "Minimax M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } }, "minimax/minimax-m2.5-highspeed": { "id": "minimax/minimax-m2.5-highspeed", "name": "MiniMax M2.5 High Speed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131000 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.03, "cache_write": 0.375 } }, "minimax/minimax-m2.1-lightning": { "id": "minimax/minimax-m2.1-lightning", "name": "MiniMax M2.1 Lightning", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-23", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 2.4, "cache_read": 0.03, "cache_write": 0.375 } }, "kwaipilot/kat-coder-air-v2.5": { "id": "kwaipilot/kat-coder-air-v2.5", "name": "Kat Coder Air V2.5", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-07-10", "last_updated": "2026-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 80000 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.03 } }, "kwaipilot/kat-coder-pro-v2.5": { "id": "kwaipilot/kat-coder-pro-v2.5", "name": "Kat Coder Pro V2.5", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-07-10", "last_updated": "2026-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 80000 }, "cost": { "input": 0.74, "output": 2.96, "cache_read": 0.15 } }, "kwaipilot/kat-coder-pro-v1": { "id": "kwaipilot/kat-coder-pro-v1", "name": "KAT-Coder-Pro V1", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-11-09", "last_updated": "2025-10-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "kwaipilot/kat-coder-pro-v2": { "id": "kwaipilot/kat-coder-pro-v2", "name": "Kat Coder Pro V2", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } } } }, "submodel": { "id": "submodel", "env": ["SUBMODEL_INSTAGEN_ACCESS_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://llm.submodel.ai/v1", "name": "submodel", "doc": "https://submodel.gitbook.io", "models": { "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen3 235B A22B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-23", "last_updated": "2025-08-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.2, "output": 0.6 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08-23", "last_updated": "2025-08-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2, "output": 0.8 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08-23", "last_updated": "2025-08-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.2, "output": 0.3 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-23", "last_updated": "2025-08-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.1, "output": 0.5 } }, "zai-org/GLM-4.5-FP8": { "id": "zai-org/GLM-4.5-FP8", "name": "GLM 4.5 FP8", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.2, "output": 0.8 } }, "zai-org/GLM-4.5-Air": { "id": "zai-org/GLM-4.5-Air", "name": "GLM 4.5 Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.1, "output": 0.5 } }, "deepseek-ai/DeepSeek-V3-0324": { "id": "deepseek-ai/DeepSeek-V3-0324", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08-23", "last_updated": "2025-08-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 75000, "output": 163840 }, "cost": { "input": 0.2, "output": 0.8 } }, "deepseek-ai/DeepSeek-R1-0528": { "id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-23", "last_updated": "2025-08-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 75000, "output": 163840 }, "cost": { "input": 0.5, "output": 2.15 } }, "deepseek-ai/DeepSeek-V3.1": { "id": "deepseek-ai/DeepSeek-V3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-23", "last_updated": "2025-08-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 75000, "output": 163840 }, "cost": { "input": 0.2, "output": 0.8 } } } }, "huggingface": { "id": "huggingface", "env": ["HF_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://router.huggingface.co/v1", "name": "Hugging Face", "doc": "https://huggingface.co/docs/inference-providers", "models": { "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 4096 }, "cost": { "input": 0.59, "output": 0.79 } }, "moonshotai/Kimi-K2-Thinking": { "id": "moonshotai/Kimi-K2-Thinking", "name": "Kimi-K2-Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "moonshotai/Kimi-K2-Instruct-0905": { "id": "moonshotai/Kimi-K2-Instruct-0905", "name": "Kimi-K2-Instruct-0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 1, "output": 3 } }, "moonshotai/Kimi-K2-Instruct": { "id": "moonshotai/Kimi-K2-Instruct", "name": "Kimi-K2-Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-07-14", "last_updated": "2025-07-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 1, "output": 3 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi-K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-01", "last_updated": "2026-01-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4 } }, "stepfun-ai/Step-3.5-Flash": { "id": "stepfun-ai/Step-3.5-Flash", "name": "Step 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3 } }, "stepfun-ai/Step-3.7-Flash": { "id": "stepfun-ai/Step-3.7-Flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 256000 }, "cost": { "input": 0.2, "output": 1.15 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.14, "output": 0.4 } }, "google/gemma-4-26B-A4B-it": { "id": "google/gemma-4-26B-A4B-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.13, "output": 0.4 } }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.15, "output": 0.95 } }, "Qwen/Qwen3-Coder-Next": { "id": "Qwen/Qwen3-Coder-Next", "name": "Qwen3-Coder-Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.2, "output": 1.5 } }, "Qwen/Qwen3-Embedding-8B": { "id": "Qwen/Qwen3-Embedding-8B", "name": "Qwen 3 Embedding 8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.01, "output": 0 } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.47, "output": 3.19 } }, "Qwen/Qwen3-Next-80B-A3B-Instruct": { "id": "Qwen/Qwen3-Next-80B-A3B-Instruct", "name": "Qwen3-Next-80B-A3B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-11", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 66536 }, "cost": { "input": 0.25, "output": 1 } }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen3.5-397B-A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.6, "output": 3.6 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen3-235B-A22B-Thinking-2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.3, "output": 3 } }, "Qwen/Qwen3-Embedding-4B": { "id": "Qwen/Qwen3-Embedding-4B", "name": "Qwen 3 Embedding 4B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 2048 }, "cost": { "input": 0.01, "output": 0 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen3-Coder-480B-A35B-Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 66536 }, "cost": { "input": 2, "output": 2 } }, "Qwen/Qwen3.5-122B-A10B": { "id": "Qwen/Qwen3.5-122B-A10B", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 3.2 } }, "Qwen/Qwen3-Coder-30B-A3B-Instruct": { "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.07, "output": 0.26 } }, "Qwen/Qwen3.5-27B": { "id": "Qwen/Qwen3.5-27B", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2.4 } }, "Qwen/Qwen3-Next-80B-A3B-Thinking": { "id": "Qwen/Qwen3-Next-80B-A3B-Thinking", "name": "Qwen3-Next-80B-A3B-Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-11", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.3, "output": 2 } }, "Qwen/Qwen3.5-9B": { "id": "Qwen/Qwen3.5-9B", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.17, "output": 0.25 } }, "Qwen/Qwen3-32B": { "id": "Qwen/Qwen3-32B", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.29, "output": 0.59 } }, "Qwen/Qwen3-235B-A22B": { "id": "Qwen/Qwen3-235B-A22B", "name": "Qwen3 235B-A22B", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 16384 }, "cost": { "input": 0.2, "output": 0.8 } }, "Qwen/Qwen3.5-35B-A3B": { "id": "Qwen/Qwen3.5-35B-A3B", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.25, "output": 2 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.25, "output": 0.69 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.1, "output": 0.5 } }, "XiaomiMiMo/MiMo-V2.5-Pro": { "id": "XiaomiMiMo/MiMo-V2.5-Pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1, "output": 3 } }, "XiaomiMiMo/MiMo-V2-Flash": { "id": "XiaomiMiMo/MiMo-V2-Flash", "name": "MiMo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 4096 }, "cost": { "input": 0.1, "output": 0.3 } }, "zai-org/GLM-4.7-Flash": { "id": "zai-org/GLM-4.7-Flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "zai-org/GLM-4.6": { "id": "zai-org/GLM-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.55, "output": 2.2 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "zai-org/GLM-4.5V": { "id": "zai-org/GLM-4.5V", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4 } }, "zai-org/GLM-4.5-Air": { "id": "zai-org/GLM-4.5-Air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.13, "output": 0.85 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-03", "last_updated": "2026-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "zai-org/GLM-4.5": { "id": "zai-org/GLM-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2 } }, "deepseek-ai/DeepSeek-R1": { "id": "deepseek-ai/DeepSeek-R1", "name": "DeepSeek-R1", "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 32768 }, "cost": { "input": 0.7, "output": 2.5 } }, "deepseek-ai/DeepSeek-R1-0528": { "id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek-R1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 3, "output": 5 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 393216 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.28, "output": 0.4 } }, "MiniMaxAI/MiniMax-M2.1": { "id": "MiniMaxAI/MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-10", "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMaxAI/MiniMax-M2": { "id": "MiniMaxAI/MiniMax-M2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "MiniMaxAI/MiniMax-M3": { "id": "MiniMaxAI/MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMaxAI/MiniMax-M2.7": { "id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } } } }, "minimax-coding-plan": { "id": "minimax-coding-plan", "env": ["MINIMAX_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://api.minimax.io/anthropic/v1", "name": "MiniMax Token Plan (minimax.io)", "doc": "https://platform.minimax.io/docs/token-plan/intro", "models": { "MiniMax-M2.1": { "id": "MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "MiniMax-M2.5-highspeed": { "id": "MiniMax-M2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2.7-highspeed": { "id": "MiniMax-M2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2": { "id": "MiniMax-M2", "name": "MiniMax-M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M3": { "id": "MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2.7": { "id": "MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "novita-ai": { "id": "novita-ai", "env": ["NOVITA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.novita.ai/openai", "name": "NovitaAI", "doc": "https://novita.ai/docs/guides/introduction", "models": { "inclusionai/ling-2.6-1t": { "id": "inclusionai/ling-2.6-1t", "name": "Ling-2.6-1T", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-23", "last_updated": "2026-06-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.06 } }, "inclusionai/ring-2.6-1t": { "id": "inclusionai/ring-2.6-1t", "name": "Ring-2.6-1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "ring", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-08", "last_updated": "2026-05-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.06 } }, "inclusionai/ling-2.6-flash": { "id": "inclusionai/ling-2.6-flash", "name": "Ling-2.6-flash", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "meta-llama/llama-3.1-8b-instruct": { "id": "meta-llama/llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-07-24", "last_updated": "2024-07-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.02, "output": 0.05 } }, "meta-llama/llama-3-70b-instruct": { "id": "meta-llama/llama-3-70b-instruct", "name": "Llama3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-04-25", "last_updated": "2024-04-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8000 }, "cost": { "input": 0.51, "output": 0.74 } }, "meta-llama/llama-4-scout-17b-16e-instruct": { "id": "meta-llama/llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-06", "last_updated": "2025-04-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.18, "output": 0.59 } }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-07", "last_updated": "2024-12-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 120000 }, "cost": { "input": 0.135, "output": 0.4 } }, "meta-llama/llama-3-8b-instruct": { "id": "meta-llama/llama-3-8b-instruct", "name": "Llama 3 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-25", "last_updated": "2024-04-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.04, "output": 0.04 } }, "meta-llama/llama-3.2-3b-instruct": { "id": "meta-llama/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32000 }, "cost": { "input": 0.03, "output": 0.05 } }, "meta-llama/llama-4-maverick-17b-128e-instruct-fp8": { "id": "meta-llama/llama-4-maverick-17b-128e-instruct-fp8", "name": "Llama 4 Maverick Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-06", "last_updated": "2025-04-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 8192 }, "cost": { "input": 0.27, "output": 0.85 } }, "moonshotai/kimi-k2-instruct": { "id": "moonshotai/kimi-k2-instruct", "name": "Kimi K2 Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-11", "last_updated": "2025-07-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.57, "output": 2.3 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-11-07", "last_updated": "2026-06-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.8, "output": 3.4, "cache_read": 0.16 } }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5 } }, "minimaxai/minimax-m1-80k": { "id": "minimaxai/minimax-m1-80k", "name": "MiniMax M1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 40000 }, "cost": { "input": 0.55, "output": 2.2 } }, "baidu/ernie-4.5-vl-28b-a3b": { "id": "baidu/ernie-4.5-vl-28b-a3b", "name": "ERNIE 4.5 VL 28B A3B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2026-06-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 30000, "output": 8000 }, "cost": { "input": 0.14, "output": 0.56 } }, "baidu/ernie-4.5-21B-a3b": { "id": "baidu/ernie-4.5-21B-a3b", "name": "ERNIE 4.5 21B A3B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "ernie", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 120000, "output": 8000 }, "cost": { "input": 0.07, "output": 0.28 } }, "baidu/ernie-4.5-vl-424b-a47b": { "id": "baidu/ernie-4.5-vl-424b-a47b", "name": "ERNIE 4.5 VL 424B A47B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 123000, "output": 16000 }, "cost": { "input": 0.42, "output": 1.25 } }, "baidu/ernie-4.5-21B-a3b-thinking": { "id": "baidu/ernie-4.5-21B-a3b-thinking", "name": "ERNIE-4.5-21B-A3B-Thinking", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "ernie", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-03", "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.07, "output": 0.28 } }, "baidu/ernie-4.5-300b-a47b-paddle": { "id": "baidu/ernie-4.5-300b-a47b-paddle", "name": "ERNIE 4.5 300B A47B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 123000, "output": 12000 }, "cost": { "input": 0.28, "output": 1.1 } }, "baidu/ernie-4.5-vl-28b-a3b-thinking": { "id": "baidu/ernie-4.5-vl-28b-a3b-thinking", "name": "ERNIE-4.5-VL-28B-A3B-Thinking", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-11-26", "last_updated": "2025-11-26", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.39, "output": 0.39 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.14, "output": 0.4 } }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.13, "output": 0.4 } }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", "name": "Gemma 3 12B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.05, "output": 0.1 } }, "google/gemma-3-27b-it": { "id": "google/gemma-3-27b-it", "name": "Gemma 3 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 98304, "output": 16384 }, "cost": { "input": 0.119, "output": 0.2 } }, "microsoft/wizardlm-2-8x22b": { "id": "microsoft/wizardlm-2-8x22b", "name": "Wizardlm 2 8x22B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-24", "last_updated": "2024-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65535, "output": 8000 }, "cost": { "input": 0.62, "output": 0.62 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "OpenAI GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-06", "last_updated": "2025-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.05, "output": 0.25 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "OpenAI: GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-08-06", "last_updated": "2025-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.04, "output": 0.15 } }, "sao10K/l31-70b-euryale-v2.2": { "id": "sao10K/l31-70b-euryale-v2.2", "name": "L31 70B Euryale V2.2", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 1.48, "output": 1.48 } }, "sao10K/L3-8B-stheno-v3.2": { "id": "sao10K/L3-8B-stheno-v3.2", "name": "L3 8B Stheno V3.2", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-11-29", "last_updated": "2024-11-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 32000 }, "cost": { "input": 0.05, "output": 0.05 } }, "sao10K/l3-8b-lunaris": { "id": "sao10K/l3-8b-lunaris", "name": "Sao10k L3 8B Lunaris\t", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-11-28", "last_updated": "2024-11-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.05, "output": 0.05 } }, "sao10K/l3-70b-euryale-v2.1": { "id": "sao10K/l3-70b-euryale-v2.1", "name": "L3 70B Euryale V2.1\t", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-06-18", "last_updated": "2024-06-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 1.48, "output": 1.48 } }, "baichuan/baichuan-m2-32b": { "id": "baichuan/baichuan-m2-32b", "name": "baichuan-m2-32b", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "baichuan", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-12", "release_date": "2025-08-13", "last_updated": "2025-08-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.07, "output": 0.07 } }, "mistralai/mistral-nemo": { "id": "mistralai/mistral-nemo", "name": "Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-07-30", "last_updated": "2024-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 60288, "output": 16000 }, "cost": { "input": 0.04, "output": 0.17 } }, "xiaomimimo/mimo-v2-flash": { "id": "xiaomimimo/mimo-v2-flash", "name": "XiaomiMiMo/MiMo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-19", "last_updated": "2025-12-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.3 } }, "xiaomimimo/mimo-v2-pro": { "id": "xiaomimimo/mimo-v2-pro", "name": "MiMo-V2-Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-05-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 2, "output": 6, "cache_read": 0.4, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "xiaomimimo/mimo-v2.5-pro": { "id": "xiaomimimo/mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-05-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.522, "output": 1.044, "cache_read": 0.0043, "tiers": [ { "input": 0.522, "output": 1.044, "cache_read": 0.0043, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.522, "output": 1.044, "cache_read": 0.0043 } } }, "gryphe/mythomax-l2-13b": { "id": "gryphe/mythomax-l2-13b", "name": "Mythomax L2 13B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-25", "last_updated": "2024-04-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 3200 }, "cost": { "input": 0.09, "output": 0.09 } }, "zai-org/glm-4.7": { "id": "zai-org/glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "zai-org/glm-4.5v": { "id": "zai-org/glm-4.5v", "name": "GLM 4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glmv", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "video", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8, "cache_read": 0.11 } }, "zai-org/glm-4.5": { "id": "zai-org/glm-4.5", "name": "GLM-4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "zai-org/autoglm-phone-9b-multilingual": { "id": "zai-org/autoglm-phone-9b-multilingual", "name": "AutoGLM-Phone-9B-Multilingual", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-12-10", "last_updated": "2025-12-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.035, "output": 0.138 } }, "zai-org/glm-5.1": { "id": "zai-org/glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1.38, "output": 4.4, "cache_read": 0.26 } }, "zai-org/glm-4.6": { "id": "zai-org/glm-4.6", "name": "GLM 4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.55, "output": 2.2, "cache_read": 0.11 } }, "zai-org/glm-5.2": { "id": "zai-org/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "zai-org/glm-4.6v": { "id": "zai-org/glm-4.6v", "name": "GLM 4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glmv", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "video", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9, "cache_read": 0.055 } }, "zai-org/glm-4.5-air": { "id": "zai-org/glm-4.5-air", "name": "GLM 4.5 Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-10-13", "last_updated": "2025-10-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.13, "output": 0.85, "cache_read": 0.025 } }, "zai-org/glm-4.7-flash": { "id": "zai-org/glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.07, "output": 0.4, "cache_read": 0.01 } }, "zai-org/glm-5": { "id": "zai-org/glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "paddlepaddle/paddleocr-vl": { "id": "paddlepaddle/paddleocr-vl", "name": "PaddleOCR-VL", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-22", "last_updated": "2025-10-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.02, "output": 0.02 } }, "qwen/qwen3-vl-235b-a22b-thinking": { "id": "qwen/qwen3-vl-235b-a22b-thinking", "name": "Qwen3 VL 235B A22B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.98, "output": 3.95 } }, "qwen/qwen3-vl-30b-a3b-thinking": { "id": "qwen/qwen3-vl-30b-a3b-thinking", "name": "qwen/qwen3-vl-30b-a3b-thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-11", "last_updated": "2025-10-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.2, "output": 1 } }, "qwen/qwen3-235b-a22b-instruct-2507": { "id": "qwen/qwen3-235b-a22b-instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.09, "output": 0.58 } }, "qwen/qwen3-coder-30b-a3b-instruct": { "id": "qwen/qwen3-coder-30b-a3b-instruct", "name": "Qwen3 Coder 30b A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-09", "last_updated": "2025-10-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 160000, "output": 32768 }, "cost": { "input": 0.07, "output": 0.27 } }, "qwen/qwen3-8b-fp8": { "id": "qwen/qwen3-8b-fp8", "name": "Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 20000 }, "cost": { "input": 0.035, "output": 0.138 } }, "qwen/qwen3-next-80b-a3b-instruct": { "id": "qwen/qwen3-next-80b-a3b-instruct", "name": "Qwen3 Next 80B A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-10", "last_updated": "2025-09-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 1.5 } }, "qwen/qwen3-vl-8b-instruct": { "id": "qwen/qwen3-vl-8b-instruct", "name": "qwen/qwen3-vl-8b-instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-17", "last_updated": "2025-10-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.08, "output": 0.5 } }, "qwen/qwen3-omni-30b-a3b-thinking": { "id": "qwen/qwen3-omni-30b-a3b-thinking", "name": "Qwen3 Omni 30B A3B Thinking", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text", "audio", "video", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.25, "output": 0.97, "input_audio": 2.2, "output_audio": 1.788 } }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen3.7-Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.25, "output": 3.75, "cache_read": 0.25, "cache_write": 1.5625 } }, "qwen/qwen3-max": { "id": "qwen/qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 2.11, "output": 8.45 } }, "qwen/qwen2.5-7b-instruct": { "id": "qwen/qwen2.5-7b-instruct", "name": "Qwen2.5 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.07, "output": 0.07 } }, "qwen/qwen3-next-80b-a3b-thinking": { "id": "qwen/qwen3-next-80b-a3b-thinking", "name": "Qwen3 Next 80B A3B Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-10", "last_updated": "2025-09-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 1.5 } }, "qwen/qwen3-235b-a22b-thinking-2507": { "id": "qwen/qwen3-235b-a22b-thinking-2507", "name": "Qwen3 235B A22b Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.3, "output": 3 } }, "qwen/qwen-mt-plus": { "id": "qwen/qwen-mt-plus", "name": "Qwen MT Plus", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-03", "last_updated": "2025-09-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 8192 }, "cost": { "input": 0.25, "output": 0.75 } }, "qwen/qwen3-32b-fp8": { "id": "qwen/qwen3-32b-fp8", "name": "Qwen3 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 20000 }, "cost": { "input": 0.1, "output": 0.45 } }, "qwen/qwen3-4b-fp8": { "id": "qwen/qwen3-4b-fp8", "name": "Qwen3 4B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 20000 }, "cost": { "input": 0.03, "output": 0.03 } }, "qwen/qwen2.5-vl-72b-instruct": { "id": "qwen/qwen2.5-vl-72b-instruct", "name": "Qwen2.5 VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.8, "output": 0.8 } }, "qwen/qwen3-30b-a3b-fp8": { "id": "qwen/qwen3-30b-a3b-fp8", "name": "Qwen3 30B A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 20000 }, "cost": { "input": 0.09, "output": 0.45 } }, "qwen/qwen3.5-27b": { "id": "qwen/qwen3.5-27b", "name": "Qwen3.5-27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2.4 } }, "qwen/qwen-2.5-72b-instruct": { "id": "qwen/qwen-2.5-72b-instruct", "name": "Qwen 2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-10-15", "last_updated": "2024-10-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 8192 }, "cost": { "input": 0.38, "output": 0.4 } }, "qwen/qwen3-coder-next": { "id": "qwen/qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.2, "output": 1.5 } }, "qwen/qwen3.5-35b-a3b": { "id": "qwen/qwen3.5-35b-a3b", "name": "Qwen3.5-35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.25, "output": 2 } }, "qwen/qwen3-coder-480b-a35b-instruct": { "id": "qwen/qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.38, "output": 1.55 } }, "qwen/qwen3.5-397b-a17b": { "id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5-397B-A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 64000 }, "cost": { "input": 0.6, "output": 3.6 } }, "qwen/qwen3-vl-30b-a3b-instruct": { "id": "qwen/qwen3-vl-30b-a3b-instruct", "name": "qwen/qwen3-vl-30b-a3b-instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-11", "last_updated": "2025-10-11", "modalities": { "input": ["text", "video", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.2, "output": 0.7 } }, "qwen/qwen3-omni-30b-a3b-instruct": { "id": "qwen/qwen3-omni-30b-a3b-instruct", "name": "Qwen3 Omni 30B A3B Instruct", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text", "video", "audio", "image"], "output": ["text", "audio"] }, "open_weights": true, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.25, "output": 0.97, "input_audio": 2.2, "output_audio": 1.788 } }, "qwen/qwen3-vl-235b-a22b-instruct": { "id": "qwen/qwen3-vl-235b-a22b-instruct", "name": "Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.3, "output": 1.5 } }, "qwen/qwen3.5-122b-a10b": { "id": "qwen/qwen3.5-122b-a10b", "name": "Qwen3.5-122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 3.2 } }, "qwen/qwen3-235b-a22b-fp8": { "id": "qwen/qwen3-235b-a22b-fp8", "name": "Qwen3 235B A22B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 20000 }, "cost": { "input": 0.2, "output": 0.8 } }, "deepseek/deepseek-r1-0528": { "id": "deepseek/deepseek-r1-0528", "name": "DeepSeek R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.7, "output": 2.5, "cache_read": 0.35 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 393216 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "deepseek/deepseek-v3.1-terminus": { "id": "deepseek/deepseek-v3.1-terminus", "name": "Deepseek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.27, "output": 1, "cache_read": 0.135 } }, "deepseek/deepseek-v3-0324": { "id": "deepseek/deepseek-v3-0324", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.27, "output": 1.12, "cache_read": 0.135 } }, "deepseek/deepseek-ocr": { "id": "deepseek/deepseek-ocr", "name": "DeepSeek-OCR", "description": "OCR model for extracting structured text from documents and screenshots", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-10-24", "last_updated": "2025-10-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.03, "output": 0.03 } }, "deepseek/deepseek-r1-distill-llama-70b": { "id": "deepseek/deepseek-r1-distill-llama-70b", "name": "DeepSeek R1 Distill LLama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.8, "output": 0.8 } }, "deepseek/deepseek-r1-turbo": { "id": "deepseek/deepseek-r1-turbo", "name": "DeepSeek R1 (Turbo)\t", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-03-05", "last_updated": "2025-03-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 16000 }, "cost": { "input": 0.7, "output": 2.5 } }, "deepseek/deepseek-prover-v2-671b": { "id": "deepseek/deepseek-prover-v2-671b", "name": "Deepseek Prover V2 671B", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 160000, "output": 160000 }, "cost": { "input": 0.7, "output": 2.5 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 393216 }, "cost": { "input": 1.6, "output": 3.2, "cache_read": 0.135 } }, "deepseek/deepseek-v3.2-exp": { "id": "deepseek/deepseek-v3.2-exp", "name": "Deepseek V3.2 Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.27, "output": 0.41 } }, "deepseek/deepseek-ocr-2": { "id": "deepseek/deepseek-ocr-2", "name": "deepseek/deepseek-ocr-2", "description": "OCR model for extracting structured text from documents and screenshots", "attachment": true, "reasoning": false, "tool_call": false, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.03, "output": 0.03 } }, "deepseek/deepseek-r1-distill-qwen-14b": { "id": "deepseek/deepseek-r1-distill-qwen-14b", "name": "DeepSeek R1 Distill Qwen 14B", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "deepseek-thinking", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.15, "output": 0.15 } }, "deepseek/deepseek-v3.1": { "id": "deepseek/deepseek-v3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.27, "output": 1, "cache_read": 0.135 } }, "deepseek/deepseek-r1-0528-qwen3-8b": { "id": "deepseek/deepseek-r1-0528-qwen3-8b", "name": "DeepSeek R1 0528 Qwen3 8B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-05-29", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0.06, "output": 0.09 } }, "deepseek/deepseek-r1-distill-qwen-32b": { "id": "deepseek/deepseek-r1-distill-qwen-32b", "name": "DeepSeek R1 Distill Qwen 32B", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "deepseek-thinking", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 32000 }, "cost": { "input": 0.3, "output": 0.3 } }, "deepseek/deepseek-v3-turbo": { "id": "deepseek/deepseek-v3-turbo", "name": "DeepSeek V3 (Turbo)\t", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-03-05", "last_updated": "2025-03-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 16000 }, "cost": { "input": 0.4, "output": 1.3 } }, "deepseek/deepseek-v3.2": { "id": "deepseek/deepseek-v3.2", "name": "Deepseek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.269, "output": 0.4, "cache_read": 0.1345 } }, "minimax/minimax-m2.7-highspeed": { "id": "minimax/minimax-m2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-05-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131100 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "Minimax M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax-M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax-m2.7", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m2.5-highspeed": { "id": "minimax/minimax-m2.5-highspeed", "name": "MiniMax M2.5 Highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax-m2.5", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131100 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.03 } }, "kwaipilot/kat-coder-pro": { "id": "kwaipilot/kat-coder-pro", "name": "Kat Coder Pro", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-05", "last_updated": "2026-01-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "nousresearch/hermes-2-pro-llama-3-8b": { "id": "nousresearch/hermes-2-pro-llama-3-8b", "name": "Hermes 2 Pro Llama 3 8B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-06-27", "last_updated": "2024-06-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.14, "output": 0.14 } } } }, "xai": { "id": "xai", "env": ["XAI_API_KEY"], "npm": "@ai-sdk/xai", "name": "xAI", "doc": "https://docs.x.ai/docs/models", "models": { "grok-4.20-multi-agent-0309": { "id": "grok-4.20-multi-agent-0309", "name": "Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "grok-4.20-0309-non-reasoning": { "id": "grok-4.20-0309-non-reasoning", "name": "Grok 4.20 (Non-Reasoning)", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "grok-4.3": { "id": "grok-4.3", "name": "Grok 4.3", "description": "xAI's Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "grok-imagine-image-quality": { "id": "grok-imagine-image-quality", "name": "Grok Imagine Image Quality", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-03", "last_updated": "2026-04-03", "modalities": { "input": ["text", "image", "pdf"], "output": ["image", "pdf"] }, "open_weights": false, "limit": { "context": 8000, "output": 0 } }, "grok-imagine-video": { "id": "grok-imagine-video", "name": "Grok Imagine Video", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-01-28", "last_updated": "2026-01-28", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["video"] }, "open_weights": false, "limit": { "context": 1024, "output": 0 } }, "grok-4.5": { "id": "grok-4.5", "name": "Grok 4.5", "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 500000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.5, "tiers": [ { "input": 4, "output": 12, "cache_read": 1, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 12, "cache_read": 1 } } }, "grok-4.20-0309-reasoning": { "id": "grok-4.20-0309-reasoning", "name": "Grok 4.20 (Reasoning)", "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "grok-imagine-image": { "id": "grok-imagine-image", "name": "Grok Imagine Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-01-28", "last_updated": "2026-01-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["image", "pdf"] }, "open_weights": false, "limit": { "context": 8000, "output": 0 } }, "grok-build-0.1": { "id": "grok-build-0.1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 4, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2, "output": 4, "cache_read": 0.4 } } } } }, "privatemode-ai": { "id": "privatemode-ai", "env": ["PRIVATEMODE_API_KEY", "PRIVATEMODE_ENDPOINT"], "npm": "@ai-sdk/openai-compatible", "api": "http://localhost:8080/v1", "name": "Privatemode AI", "doc": "https://docs.privatemode.ai/api/overview", "models": { "qwen3-embedding-4b": { "id": "qwen3-embedding-4b", "name": "Qwen3-Embedding 4B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2025-06", "release_date": "2025-06-06", "last_updated": "2025-06-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 2560 }, "cost": { "input": 0, "output": 0 } }, "gemma-3-27b": { "id": "gemma-3-27b", "name": "Gemma 3 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "gpt-oss-120b", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-04", "last_updated": "2025-08-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "whisper-large-v3": { "id": "whisper-large-v3", "name": "Whisper large-v3", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-09", "release_date": "2023-09-01", "last_updated": "2023-09-01", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "qwen3-coder-30b-a3b": { "id": "qwen3-coder-30b-a3b", "name": "Qwen3-Coder 30B-A3B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } } } }, "drun": { "id": "drun", "env": ["DRUN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://chat.d.run/v1", "name": "D.Run (China)", "doc": "https://www.d.run", "models": { "public/deepseek-v3": { "id": "public/deepseek-v3", "name": "DeepSeek V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-12-26", "last_updated": "2024-12-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.28, "output": 1.1 } }, "public/deepseek-r1": { "id": "public/deepseek-r1", "name": "DeepSeek R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32000 }, "cost": { "input": 0.55, "output": 2.2 } }, "public/minimax-m25": { "id": "public/minimax-m25", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "temperature": true, "release_date": "2025-03-01", "last_updated": "2025-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.29, "output": 1.16 } } } }, "alibaba-token-plan-cn": { "id": "alibaba-token-plan-cn", "env": ["ALIBABA_TOKEN_PLAN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1", "name": "Alibaba Token Plan (China)", "doc": "https://www.alibabacloud.com/help/zh/model-studio/token-plan-overview", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "wan2.7-image-pro": { "id": "wan2.7-image-pro", "name": "Wan2.7 Image Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen-image-2.0": { "id": "qwen-image-2.0", "name": "Qwen Image 2.0", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "input": 196601, "output": 24576 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen-image-2.0-pro": { "id": "qwen-image-2.0-pro", "name": "Qwen Image 2.0 Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "wan2.7-image": { "id": "wan2.7-image", "name": "Wan2.7 Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-03", "last_updated": "2025-12-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "moonshotai": { "id": "moonshotai", "env": ["MOONSHOT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.moonshot.ai/v1", "name": "Moonshot AI", "doc": "https://platform.moonshot.ai/docs/api/chat", "models": { "kimi-k2-0905-preview": { "id": "kimi-k2-0905-preview", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "kimi-k2-thinking-turbo": { "id": "kimi-k2-thinking-turbo", "name": "Kimi K2 Thinking Turbo", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.15, "output": 8, "cache_read": 0.15 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "kimi-k2-0711-preview": { "id": "kimi-k2-0711-preview", "name": "Kimi K2 0711", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-07-14", "last_updated": "2025-07-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "kimi-k2-turbo-preview": { "id": "kimi-k2-turbo-preview", "name": "Kimi K2 Turbo", "description": "Fast Kimi model for responsive chat, coding help, and agent loops", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 2.4, "output": 10, "cache_read": 0.6 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "kimi-k2.7-code-highspeed": { "id": "kimi-k2.7-code-highspeed", "name": "Kimi K2.7 Code HighSpeed", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.9, "output": 8, "cache_read": 0.38 } } } }, "fireworks-ai": { "id": "fireworks-ai", "env": ["FIREWORKS_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.fireworks.ai/inference/v1/", "name": "Fireworks AI", "doc": "https://fireworks.ai/docs/", "models": { "accounts/fireworks/routers/kimi-k2p6-turbo": { "id": "accounts/fireworks/routers/kimi-k2p6-turbo", "name": "Kimi K2.6 Turbo", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.3 } }, "accounts/fireworks/routers/glm-5p2-fast": { "id": "accounts/fireworks/routers/glm-5p2-fast", "name": "GLM 5.2 Fast", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-26", "last_updated": "2026-06-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048575, "output": 131072 }, "cost": { "input": 2.1, "output": 6.6, "cache_read": 0.21 } }, "accounts/fireworks/routers/kimi-k2p7-code-fast": { "id": "accounts/fireworks/routers/kimi-k2p7-code-fast", "name": "Kimi K2.7 Code Fast", "description": "Kimi coding model for software agents, refactors, and repository reasoning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 1.9, "output": 8, "cache_read": 0.38 } }, "accounts/fireworks/routers/glm-5p1-fast": { "id": "accounts/fireworks/routers/glm-5p1-fast", "name": "GLM 5.1 Fast", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 131072 }, "cost": { "input": 2.8, "output": 8.8, "cache_read": 0.52 } }, "accounts/fireworks/routers/kimi-k2p6-fast": { "id": "accounts/fireworks/routers/kimi-k2p6-fast", "name": "Kimi K2.6 Fast", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-06-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.3 } }, "accounts/fireworks/models/deepseek-v4-flash": { "id": "accounts/fireworks/models/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "accounts/fireworks/models/deepseek-v4-pro": { "id": "accounts/fireworks/models/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.145 } }, "accounts/fireworks/models/minimax-m2p7": { "id": "accounts/fireworks/models/minimax-m2p7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-12", "last_updated": "2026-04-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "accounts/fireworks/models/minimax-m3": { "id": "accounts/fireworks/models/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 512000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "accounts/fireworks/models/kimi-k2p6": { "id": "accounts/fireworks/models/kimi-k2p6", "name": "Kimi K2.6", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "accounts/fireworks/models/qwen3p7-plus": { "id": "accounts/fireworks/models/qwen3p7-plus", "name": "Qwen 3.7 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1 } ], "tool_call": true, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.08 } }, "accounts/fireworks/models/glm-5p1": { "id": "accounts/fireworks/models/glm-5p1", "name": "GLM 5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "accounts/fireworks/models/glm-5p2": { "id": "accounts/fireworks/models/glm-5p2", "name": "GLM 5.2", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-16", "last_updated": "2026-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048575, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.14 } }, "accounts/fireworks/models/gpt-oss-120b": { "id": "accounts/fireworks/models/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2026-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015 } }, "accounts/fireworks/models/gpt-oss-20b": { "id": "accounts/fireworks/models/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.07, "output": 0.3, "cache_read": 0.035 } }, "accounts/fireworks/models/kimi-k2p7-code": { "id": "accounts/fireworks/models/kimi-k2p7-code", "name": "Kimi K2.7 Code", "description": "Kimi coding model for software agents, refactors, and repository reasoning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } } } }, "vultr": { "id": "vultr", "env": ["VULTR_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.vultrinference.com/v1", "name": "Vultr", "doc": "https://api.vultrinference.com/", "models": { "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2 } }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2 } }, "XiaomiMiMo/MiMo-V2.5-Pro": { "id": "XiaomiMiMo/MiMo-V2.5-Pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.55, "output": 1.65 } }, "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16": { "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16", "name": "NVIDIA Nemotron 3 Nano Omni", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.13, "output": 0.38 } }, "nvidia/DeepSeek-V3.2-NVFP4": { "id": "nvidia/DeepSeek-V3.2-NVFP4", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.55, "output": 1.65 } }, "nvidia/Nemotron-Cascade-2-30B-A3B": { "id": "nvidia/Nemotron-Cascade-2-30B-A3B", "name": "NVIDIA Nemotron Cascade 2", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.15, "output": 0.6 } }, "zai-org/GLM-5.2-FP8": { "id": "zai-org/GLM-5.2-FP8", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 393216, "output": 131072 }, "cost": { "input": 0.85, "output": 3.1 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.3, "output": 1 } }, "MiniMaxAI/MiniMax-M2.7": { "id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } } } }, "302ai": { "id": "302ai", "env": ["302AI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.302.ai/v1", "name": "302.AI", "doc": "https://doc.302.ai", "models": { "gpt-5.4-mini-2026-03-17": { "id": "gpt-5.4-mini-2026-03-17", "name": "gpt-5.4-mini-2026-03-17", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5 } }, "chatgpt-4o-latest": { "id": "chatgpt-4o-latest", "name": "chatgpt-4o-latest", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-09", "release_date": "2024-08-08", "last_updated": "2024-08-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 5, "output": 15 } }, "gpt-5.4-nano-2026-03-17": { "id": "gpt-5.4-nano-2026-03-17", "name": "gpt-5.4-nano-2026-03-17", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25 } }, "kimi-k2-0905-preview": { "id": "kimi-k2-0905-preview", "name": "kimi-k2-0905-preview", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.632, "output": 2.53 } }, "grok-4.20-beta-0309-non-reasoning": { "id": "grok-4.20-beta-0309-non-reasoning", "name": "grok-4.20-beta-0309-non-reasoning", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 2, "output": 6 } }, "gemini-2.5-flash-nothink": { "id": "gemini-2.5-flash-nothink", "name": "gemini-2.5-flash-nothink", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-24", "last_updated": "2025-06-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5 } }, "qwen-plus": { "id": "qwen-plus", "name": "Qwen-Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.12, "output": 1.2 } }, "glm-4.7": { "id": "glm-4.7", "name": "glm-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.286, "output": 1.142 } }, "qwen3-235b-a22b-instruct-2507": { "id": "qwen3-235b-a22b-instruct-2507", "name": "qwen3-235b-a22b-instruct-2507", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 65536 }, "cost": { "input": 0.29, "output": 1.143 } }, "glm-4.5v": { "id": "glm-4.5v", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 16384 }, "cost": { "input": 0.29, "output": 0.86 } }, "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "claude-opus-4-5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "gemini-2.5-pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.25, "output": 10 } }, "gpt-5": { "id": "gpt-5", "name": "gpt-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "claude-haiku-4-5-20251001", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-16", "last_updated": "2025-10-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5 } }, "kimi-k2-thinking-turbo": { "id": "kimi-k2-thinking-turbo", "name": "kimi-k2-thinking-turbo", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.265, "output": 9.119 } }, "claude-3-5-haiku-20241022": { "id": "claude-3-5-haiku-20241022", "name": "claude-3-5-haiku-20241022", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 0.8, "output": 4 } }, "glm-4.5": { "id": "glm-4.5", "name": "GLM-4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.286, "output": 1.142 } }, "gpt-5-pro": { "id": "gpt-5-pro", "name": "gpt-5-pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-08", "last_updated": "2025-10-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "grok-4.20-beta-0309-reasoning": { "id": "grok-4.20-beta-0309-reasoning", "name": "grok-4.20-beta-0309-reasoning", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 2, "output": 6 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "gemini-2.5-flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5 } }, "gpt-4o": { "id": "gpt-4o", "name": "gpt-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "MiniMax-M2.1": { "id": "MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-19", "last_updated": "2025-12-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "gemini-2.5-flash-lite-preview-09-2025": { "id": "gemini-2.5-flash-lite-preview-09-2025", "name": "gemini-2.5-flash-lite-preview-09-2025", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-26", "last_updated": "2025-09-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4 } }, "doubao-seed-1-6-vision-250815": { "id": "doubao-seed-1-6-vision-250815", "name": "doubao-seed-1-6-vision-250815", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0.114, "output": 1.143 } }, "claude-opus-4-1-20250805": { "id": "claude-opus-4-1-20250805", "name": "claude-opus-4-1-20250805", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75 } }, "qwen3-max-2025-09-23": { "id": "qwen3-max-2025-09-23", "name": "qwen3-max-2025-09-23", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-24", "last_updated": "2025-09-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 258048, "output": 65536 }, "cost": { "input": 0.86, "output": 3.43 } }, "glm-4.7-flashx": { "id": "glm-4.7-flashx", "name": "glm-4.7-flashx", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-20", "last_updated": "2026-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.0715, "output": 0.429 } }, "glm-5.1": { "id": "glm-5.1", "name": "glm-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-10", "last_updated": "2026-04-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.86, "output": 3.5 } }, "glm-4.6": { "id": "glm-4.6", "name": "glm-4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.286, "output": 1.142 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "kimi-k2-thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.575, "output": 2.3 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "claude-sonnet-4-5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "glm-4.5-x": { "id": "glm-4.5-x", "name": "glm-4.5-x", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.143, "output": 2.29 } }, "deepseek-v3.2-thinking": { "id": "deepseek-v3.2-thinking", "name": "DeepSeek-V3.2-Thinking", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.29, "output": 0.43 } }, "claude-sonnet-4-6-thinking": { "id": "claude-sonnet-4-6-thinking", "name": "claude-sonnet-4-6-thinking", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2026-02-18", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "claude-opus-4-7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-31", "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "grok-4-1-fast-non-reasoning": { "id": "grok-4-1-fast-non-reasoning", "name": "grok-4-1-fast-non-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "gpt-5.4-nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25 } }, "claude-opus-4-1-20250805-thinking": { "id": "claude-opus-4-1-20250805-thinking", "name": "claude-opus-4-1-20250805-thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-05-27", "last_updated": "2025-05-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75 } }, "glm-4.5-airx": { "id": "glm-4.5-airx", "name": "glm-4.5-airx", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.572, "output": 1.714 } }, "grok-4.1": { "id": "grok-4.1", "name": "grok-4.1", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 2, "output": 10 } }, "gemini-2.5-flash-preview-09-2025": { "id": "gemini-2.5-flash-preview-09-2025", "name": "gemini-2.5-flash-preview-09-2025", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-26", "last_updated": "2025-09-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5 } }, "claude-opus-4-5-20251101": { "id": "claude-opus-4-5-20251101", "name": "claude-opus-4-5-20251101", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25 } }, "claude-opus-4-20250514": { "id": "claude-opus-4-20250514", "name": "claude-opus-4-20250514", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75 } }, "gemini-3-pro-image-preview": { "id": "gemini-3-pro-image-preview", "name": "gemini-3-pro-image-preview", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-06", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 64000 }, "cost": { "input": 2, "output": 120 } }, "gemini-2.5-flash-image": { "id": "gemini-2.5-flash-image", "name": "gemini-2.5-flash-image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-10-08", "last_updated": "2025-10-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.3, "output": 30 } }, "glm-for-coding": { "id": "glm-for-coding", "name": "glm-for-coding", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.086, "output": 0.343 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "gpt-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-12", "last_updated": "2025-12-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "qwen-flash": { "id": "qwen-flash", "name": "Qwen-Flash", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.022, "output": 0.22 } }, "claude-opus-4-6-thinking": { "id": "claude-opus-4-6-thinking", "name": "claude-opus-4-6-thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-02-06", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "claude-sonnet-4-20250514": { "id": "claude-sonnet-4-20250514", "name": "claude-sonnet-4-20250514", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "gpt-5.1-chat-latest": { "id": "gpt-5.1-chat-latest", "name": "gpt-5.1-chat-latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10 } }, "gpt-5.2-chat-latest": { "id": "gpt-5.2-chat-latest", "name": "gpt-5.2-chat-latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-12", "last_updated": "2025-12-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14 } }, "grok-4-fast-reasoning": { "id": "grok-4-fast-reasoning", "name": "grok-4-fast-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "gpt-4.1-nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4 } }, "claude-sonnet-4-5-20250929-thinking": { "id": "claude-sonnet-4-5-20250929-thinking", "name": "claude-sonnet-4-5-20250929-thinking", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "MiniMax-M2.7-highspeed": { "id": "MiniMax-M2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 4.8 } }, "MiniMax-M2": { "id": "MiniMax-M2", "name": "MiniMax-M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-10-26", "last_updated": "2025-10-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.33, "output": 1.32 } }, "gemini-3.1-flash-image-preview": { "id": "gemini-3.1-flash-image-preview", "name": "gemini-3.1-flash-image-preview", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-27", "last_updated": "2026-02-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 60 } }, "qwen3-235b-a22b": { "id": "qwen3-235b-a22b", "name": "Qwen3-235B-A22B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.29, "output": 2.86 } }, "ministral-14b-2512": { "id": "ministral-14b-2512", "name": "ministral-14b-2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.33, "output": 0.33 } }, "glm-4.6v": { "id": "glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.145, "output": 0.43 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "claude-haiku-4-5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-16", "last_updated": "2025-10-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "gpt-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 0, "tiers": [ { "input": 5, "output": 22.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5 } } }, "doubao-seed-1-6-thinking-250715": { "id": "doubao-seed-1-6-thinking-250715", "name": "doubao-seed-1-6-thinking-250715", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-07-15", "last_updated": "2025-07-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 16000 }, "cost": { "input": 0.121, "output": 1.21 } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "gpt-5.4-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "gpt-4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8 } }, "doubao-seed-code-preview-251028": { "id": "doubao-seed-code-preview-251028", "name": "doubao-seed-code-preview-251028", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-11-11", "last_updated": "2025-11-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0.17, "output": 1.14 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "claude-opus-4-6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-06", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "qwen3-coder-480b-a35b-instruct": { "id": "qwen3-coder-480b-a35b-instruct", "name": "qwen3-coder-480b-a35b-instruct", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.86, "output": 3.43 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "claude-sonnet-4-5-20250929", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "deepseek-reasoner": { "id": "deepseek-reasoner", "name": "Deepseek-Reasoner", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.29, "output": 0.43 } }, "grok-4-1-fast-reasoning": { "id": "grok-4-1-fast-reasoning", "name": "grok-4-1-fast-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5 } }, "gemini-3-pro-preview": { "id": "gemini-3-pro-preview", "name": "gemini-3-pro-preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 2, "output": 12 } }, "gpt-5-thinking": { "id": "gpt-5-thinking", "name": "gpt-5-thinking", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "gpt-5-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "gpt-4.1-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6 } }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.72, "output": 3.2 } }, "gpt-5.4-pro": { "id": "gpt-5.4-pro", "name": "gpt-5.4-pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "cache_read": 0, "cache_write": 0, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "glm-4.5-air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.1143, "output": 0.286 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "claude-sonnet-4-6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-18", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "grok-4-fast-non-reasoning": { "id": "grok-4-fast-non-reasoning", "name": "grok-4-fast-non-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "gemini-3-flash-preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-12-18", "last_updated": "2025-12-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3 } }, "deepseek-chat": { "id": "deepseek-chat", "name": "Deepseek-Chat", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-11-29", "last_updated": "2024-11-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.29, "output": 0.43 } }, "MiniMax-M1": { "id": "MiniMax-M1", "name": "MiniMax-M1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-16", "last_updated": "2025-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.132, "output": 1.254 } }, "grok-4.20-multi-agent-beta-0309": { "id": "grok-4.20-multi-agent-beta-0309", "name": "grok-4.20-multi-agent-beta-0309", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 2, "output": 6 } }, "glm-5": { "id": "glm-5", "name": "glm-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.6 } }, "qwen-max-latest": { "id": "qwen-max-latest", "name": "Qwen-Max-Latest", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-04-03", "last_updated": "2025-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.343, "output": 1.372 } }, "mistral-large-2512": { "id": "mistral-large-2512", "name": "mistral-large-2512", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 262144 }, "cost": { "input": 1.1, "output": 3.3 } }, "MiniMax-M2.7": { "id": "MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "claude-3-5-haiku-latest": { "id": "claude-3-5-haiku-latest", "name": "claude-3-5-haiku-latest", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 0.8, "output": 4 } }, "qwen3-30b-a3b": { "id": "qwen3-30b-a3b", "name": "Qwen3-30B-A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.11, "output": 1.08 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "deepseek-v3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.29, "output": 0.43 } }, "doubao-seed-1-8-251215": { "id": "doubao-seed-1-8-251215", "name": "doubao-seed-1-8-251215", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-18", "last_updated": "2025-12-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 224000, "output": 64000 }, "cost": { "input": 0.114, "output": 0.286 } }, "claude-opus-4-5-20251101-thinking": { "id": "claude-opus-4-5-20251101-thinking", "name": "claude-opus-4-5-20251101-thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25 } }, "gemini-2.0-flash-lite": { "id": "gemini-2.0-flash-lite", "name": "gemini-2.0-flash-lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-11", "release_date": "2025-06-16", "last_updated": "2025-06-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 8192 }, "cost": { "input": 0.075, "output": 0.3 } }, "glm-5-turbo": { "id": "glm-5-turbo", "name": "glm-5-turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.72, "output": 3.2 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "gpt-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } } } }, "trustedrouter": { "id": "trustedrouter", "env": ["TRUSTEDROUTER_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.trustedrouter.com/v1", "name": "TrustedRouter", "doc": "https://trustedrouter.com/docs", "models": { "zdr": { "id": "zdr", "name": "Zero Data Retention", "description": "TrustedRouter privacy routing alias that prefers zero data retention model endpoints.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 } }, "e2e": { "id": "e2e", "name": "End-to-End Encrypted", "description": "TrustedRouter privacy routing alias for end-to-end encrypted provider routes where available.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 } }, "synth-code": { "id": "synth-code", "name": "Synth Code", "description": "TrustedRouter code synthesis orchestration alias that combines multiple model responses into one answer.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-20", "last_updated": "2026-06-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 } }, "fast": { "id": "fast", "name": "Fast", "description": "TrustedRouter speed routing alias that prefers low-latency healthy model endpoints.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 } }, "synth": { "id": "synth", "name": "Synth", "description": "TrustedRouter synthesis orchestration alias that combines multiple model responses into one answer.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-20", "last_updated": "2026-06-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 } }, "auto": { "id": "auto", "name": "Auto", "description": "TrustedRouter automatic routing alias that chooses a healthy supported model endpoint for the request.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-01", "last_updated": "2026-06-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 } }, "cheap": { "id": "cheap", "name": "Cheap", "description": "TrustedRouter low-cost routing alias that prefers inexpensive healthy model endpoints.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-01", "last_updated": "2026-06-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 } } } }, "zhipuai": { "id": "zhipuai", "env": ["ZHIPU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://open.bigmodel.cn/api/paas/v4", "name": "Zhipu AI", "doc": "https://docs.z.ai/guides/overview/pricing", "models": { "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0 } }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 5, "output": 22, "cache_read": 1.2, "cache_write": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2, "cache_write": 0 } }, "glm-4.5-flash": { "id": "glm-4.5-flash", "name": "GLM-4.5-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.7-flash": { "id": "glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.2, "output": 1.1, "cache_read": 0.03, "cache_write": 0 } }, "glm-4.6v": { "id": "glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9 } }, "glm-4.6": { "id": "glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "glm-4.7-flashx": { "id": "glm-4.7-flashx", "name": "GLM-4.7-FlashX", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.07, "output": 0.4, "cache_read": 0.01, "cache_write": 0 } }, "glm-4.5": { "id": "glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "glm-4.5v": { "id": "glm-4.5v", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } } } }, "cortecs": { "id": "cortecs", "env": ["CORTECS_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.cortecs.ai/v1", "name": "Cortecs", "doc": "https://api.cortecs.ai/v1/models", "models": { "deepseek-r1-0528": { "id": "deepseek-r1-0528", "name": "DeepSeek R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.585, "output": 2.307 } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 0.133, "output": 0.266, "cache_read": 0.0028 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 }, "cost": { "input": 0.32, "output": 1.18 } }, "deepseek-v3-0324": { "id": "deepseek-v3-0324", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.551, "output": 1.654 } }, "claude-opus4-7": { "id": "claude-opus4-7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5.6, "output": 27.99, "cache_read": 0.56, "cache_write": 6.99 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM 4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 198000 }, "cost": { "input": 0.45, "output": 2.23 } }, "qwen3-235b-a22b-instruct-2507": { "id": "qwen3-235b-a22b-instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.062, "output": 0.408 } }, "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3 Coder 30B A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.053, "output": 0.222 } }, "minimax-m2.1": { "id": "minimax-m2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196000, "output": 196000 }, "cost": { "input": 0.34, "output": 1.34 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.099, "output": 0.33 } }, "claude-4-6-sonnet": { "id": "claude-4-6-sonnet", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 3.59, "output": 17.92 } }, "claude-sonnet-4": { "id": "claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3.307, "output": 16.536 } }, "llama-4-maverick": { "id": "llama-4-maverick", "name": "Llama 4 Maverick 17B Instruct", "description": "Open multimodal Llama for strong reasoning with efficient everyday serving", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.124, "output": 0.603, "cache_read": 0.03, "cache_write": 0.151 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-03-20", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 1.654, "output": 11.024 } }, "claude-4-5-sonnet": { "id": "claude-4-5-sonnet", "name": "Claude 4.5 Sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 200000 }, "cost": { "input": 3.259, "output": 16.296 } }, "mixtral-8x7B-instruct-v0.1": { "id": "mixtral-8x7B-instruct-v0.1", "name": "Mixtral 8x7B Instruct v0.1", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2023-09", "release_date": "2023-12-11", "last_updated": "2023-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.438, "output": 0.68 } }, "glm-4.5": { "id": "glm-4.5", "name": "GLM 4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.67, "output": 2.46 } }, "kimi-k2-instruct": { "id": "kimi-k2-instruct", "name": "Kimi K2 Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-07-11", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.551, "output": 2.646 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Llama 3.3 70B Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.089, "output": 0.275 } }, "claude-opus4-8": { "id": "claude-opus4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5.64, "output": 28.198, "cache_read": 0.563, "cache_write": 7.049 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.28, "output": 4.63, "cache_read": 0.32 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-14", "last_updated": "2026-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1.31, "output": 4.1, "cache_read": 0.24 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 1.553, "output": 3.106, "cache_read": 0.003625 } }, "qwen3-next-80b-a3b-thinking": { "id": "qwen3-next-80b-a3b-thinking", "name": "Qwen3 Next 80B A3B Thinking", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-11", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.164, "output": 1.311 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.656, "output": 2.731 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.44, "output": 4.53, "cache_read": 0.39 } }, "minimax-m3": { "id": "minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 128000 }, "cost": { "input": 0.355, "output": 1.775, "cache_read": 0.089 } }, "minimax-m2": { "id": "minimax-m2", "name": "MiniMax-M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-11", "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 400000, "output": 400000 }, "cost": { "input": 0.39, "output": 1.57 } }, "minimax-m2.7": { "id": "minimax-m2.7", "name": "MiniMax-m2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 196072 }, "cost": { "input": 0.47, "output": 1.4 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.55, "output": 2.76 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT Oss 120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-01", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "qwen-2.5-72b-instruct": { "id": "qwen-2.5-72b-instruct", "name": "Qwen2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 33000, "output": 33000 }, "cost": { "input": 0.062, "output": 0.231 } }, "hermes-4-70b": { "id": "hermes-4-70b", "name": "Hermes 4 70B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.116, "output": 0.358 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 200000 }, "cost": { "input": 1.09, "output": 5.43 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 3, "output": 16.13, "cache_read": 0.25 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.81, "output": 3.54, "cache_read": 0.2 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT 4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2.354, "output": 9.417 } }, "nova-pro-v1": { "id": "nova-pro-v1", "name": "Nova Pro 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova-pro", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 5000 }, "cost": { "input": 1.016, "output": 4.061 } }, "qwen3-coder-next": { "id": "qwen3-coder-next", "name": "Qwen3 Coder Next 80B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-04", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.158, "output": 0.84 } }, "claude-opus4-5": { "id": "claude-opus4-5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 200000 }, "cost": { "input": 5.98, "output": 29.89 } }, "qwen3-coder-480b-a35b-instruct": { "id": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.441, "output": 1.984 } }, "devstral-2512": { "id": "devstral-2512", "name": "Devstral 2 2512", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0, "output": 0 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen3.5 397B A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 250000, "output": 250000 }, "cost": { "input": 0.6, "output": 3.6 } }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.235, "output": 4.118, "cache_read": 0.308, "cache_write": 1.544 } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "GLM 4.5 Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-01", "last_updated": "2025-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.22, "output": 1.34 } }, "glm-4.7-flash": { "id": "glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 203000, "output": 203000 }, "cost": { "input": 0.09, "output": 0.53 } }, "intellect-3": { "id": "intellect-3", "name": "INTELLECT 3", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-26", "last_updated": "2025-11-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.219, "output": 1.202 } }, "llama-3.1-405b-instruct": { "id": "llama-3.1-405b-instruct", "name": "Llama 3.1 405B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM 5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "cost": { "input": 1.08, "output": 3.44 } }, "mistral-large-2512": { "id": "mistral-large-2512", "name": "Mistral Large 3 2512", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 0.05 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.266, "output": 0.444 } }, "codestral-2508": { "id": "codestral-2508", "name": "Codestral 2508", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.3, "output": 0.9, "cache_read": 0.03 } }, "qwen3.5-122b-a10b": { "id": "qwen3.5-122b-a10b", "name": "Qwen3.5 122B A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.444, "output": 3.106 } }, "nemotron-3-super-120b-a12b": { "id": "nemotron-3-super-120b-a12b", "name": "Nemotron 3 Super 120B A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.266, "output": 0.799 } }, "glm-5-turbo": { "id": "glm-5-turbo", "name": "GLM-5-Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.235, "output": 4.118, "cache_read": 0.308, "cache_write": 1.544 } }, "claude-opus4-6": { "id": "claude-opus4-6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 5.98, "output": 29.89 } } } }, "nebius": { "id": "nebius", "env": ["NEBIUS_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.tokenfactory.nebius.com/v1", "name": "Nebius Token Factory", "doc": "https://docs.tokenfactory.nebius.com/", "models": { "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama-3.3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-12-05", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 8192 }, "cost": { "input": 0.13, "output": 0.4, "cache_read": 0.013, "cache_write": 0.16 } }, "moonshotai/Kimi-K2.5-fast": { "id": "moonshotai/Kimi-K2.5-fast", "name": "Kimi-K2.5-fast", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-12-15", "last_updated": "2026-02-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.5, "output": 2.5, "cache_read": 0.05, "cache_write": 0.625 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi-K2.5", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-12-15", "last_updated": "2026-02-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.5, "output": 2.5, "reasoning": 2.5, "cache_read": 0.05, "cache_write": 0.625 } }, "google/gemma-3-27b-it": { "id": "google/gemma-3-27b-it", "name": "Gemma-3-27b-it", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-10", "release_date": "2026-01-20", "last_updated": "2026-02-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 110000, "input": 100000, "output": 8192 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01, "cache_write": 0.125 } }, "Qwen/Qwen3-Next-80B-A3B-Thinking-fast": { "id": "Qwen/Qwen3-Next-80B-A3B-Thinking-fast", "name": "Qwen3-Next-80B-A3B-Thinking-fast", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-25", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "input": 7000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.15, "output": 1.2, "cache_read": 0.015, "cache_write": 0.1875 } }, "Qwen/Qwen2.5-VL-72B-Instruct": { "id": "Qwen/Qwen2.5-VL-72B-Instruct", "name": "Qwen2.5-VL-72B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-20", "last_updated": "2026-02-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 8192 }, "cost": { "input": 0.25, "output": 0.75, "cache_read": 0.025, "cache_write": 0.31 } }, "Qwen/Qwen3-Embedding-8B": { "id": "Qwen/Qwen3-Embedding-8B", "name": "Qwen3-Embedding-8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "knowledge": "2025-10", "release_date": "2026-01-10", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "input": 32768, "output": 0 }, "cost": { "input": 0.01, "output": 0 } }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen3.5-397B-A17B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-15", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "input": 250000, "output": 8192 }, "cost": { "input": 0.6, "output": 3.6, "cache_read": 0.06, "cache_write": 0.75 } }, "Qwen/Qwen3.5-397B-A17B-fast": { "id": "Qwen/Qwen3.5-397B-A17B-fast", "name": "Qwen3.5-397B-A17B-fast", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-15", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "input": 7000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.6, "output": 3.6, "cache_read": 0.06, "cache_write": 0.75 } }, "Qwen/Qwen3-30B-A3B-Instruct-2507": { "id": "Qwen/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen3-30B-A3B-Instruct-2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-01-28", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 8192 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01, "cache_write": 0.125 } }, "Qwen/Qwen3-Next-80B-A3B-Thinking": { "id": "Qwen/Qwen3-Next-80B-A3B-Thinking", "name": "Qwen3-Next-80B-A3B-Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-01-28", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 16384 }, "cost": { "input": 0.15, "output": 1.2, "reasoning": 1.2, "cache_read": 0.015, "cache_write": 0.18 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-25", "last_updated": "2025-10-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.2, "output": 0.6 } }, "Qwen/Qwen3-32B": { "id": "Qwen/Qwen3-32B", "name": "Qwen3-32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-01-28", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 8192 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01, "cache_write": 0.125 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507-fast": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507-fast", "name": "Qwen3-235B-A22B-Thinking-2507-fast", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-25", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "input": 7000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.5, "output": 2, "cache_read": 0.05, "cache_write": 0.625 } }, "openai/gpt-oss-120b-fast": { "id": "openai/gpt-oss-120b-fast", "name": "gpt-oss-120b-fast", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-06-10", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "input": 7000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.5, "cache_read": 0.01, "cache_write": 0.125 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-09", "release_date": "2026-01-10", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 124000, "output": 8192 }, "cost": { "input": 0.15, "output": 0.6, "reasoning": 0.6, "cache_read": 0.015, "cache_write": 0.18 } }, "NousResearch/Hermes-4-405B": { "id": "NousResearch/Hermes-4-405B", "name": "Hermes-4-405B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-11", "release_date": "2026-01-30", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 8192 }, "cost": { "input": 1, "output": 3, "reasoning": 3, "cache_read": 0.1, "cache_write": 1.25 } }, "NousResearch/Hermes-4-70B": { "id": "NousResearch/Hermes-4-70B", "name": "Hermes-4-70B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-11", "release_date": "2026-01-30", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 8192 }, "cost": { "input": 0.13, "output": 0.4, "reasoning": 0.4, "cache_read": 0.013, "cache_write": 0.16 } }, "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B": { "id": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B", "name": "Nemotron-3-Nano-30B-A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-08-10", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "input": 30000, "output": 4096 }, "cost": { "input": 0.06, "output": 0.24, "cache_read": 0.006, "cache_write": 0.075 } }, "nvidia/Nemotron-3-Nano-Omni": { "id": "nvidia/Nemotron-3-Nano-Omni", "name": "Nemotron-3-Nano-Omni", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-20", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "input": 60000, "output": 8192 }, "cost": { "input": 0.06, "output": 0.24, "cache_read": 0.006, "cache_write": 0.075 } }, "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1": { "id": "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1", "name": "Llama-3.1-Nemotron-Ultra-253B-v1", "description": "Flagship Nemotron model for high-throughput reasoning and complex agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-15", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 4096 }, "cost": { "input": 0.6, "output": 1.8, "cache_read": 0.06, "cache_write": 0.75 } }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nemotron-3-Super-120B-A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02", "release_date": "2026-03-11", "last_updated": "2026-03-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "GLM-5", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-03-01", "last_updated": "2026-03-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 16384 }, "status": "deprecated", "cost": { "input": 1, "output": 3.2, "cache_read": 0.1, "cache_write": 1 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 432000, "output": 432000 }, "cost": { "input": 1.4, "output": 4.4 } }, "deepseek-ai/DeepSeek-V3.2-fast": { "id": "deepseek-ai/DeepSeek-V3.2-fast", "name": "DeepSeek-V3.2-fast", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-27", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "input": 7000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2, "cache_read": 0.04, "cache_write": 0.5 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.75, "output": 3.5, "cache_read": 0.15 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek-V3.2", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-11", "release_date": "2026-01-20", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163000, "input": 160000, "output": 16384 }, "status": "deprecated", "cost": { "input": 0.3, "output": 0.45, "reasoning": 0.45, "cache_read": 0.03, "cache_write": 0.375 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-20", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "input": 190000, "output": 8192 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "MiniMaxAI/MiniMax-M2.5-fast": { "id": "MiniMaxAI/MiniMax-M2.5-fast", "name": "MiniMax-M2.5-fast", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-20", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "input": 7000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "PrimeIntellect/INTELLECT-3": { "id": "PrimeIntellect/INTELLECT-3", "name": "INTELLECT-3", "description": "Legacy model retained for compatibility with older integrations", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-10", "release_date": "2026-01-25", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "input": 120000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.2, "output": 1.1, "cache_read": 0.02, "cache_write": 0.25 } } } }, "auriko": { "id": "auriko", "env": ["AURIKO_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.auriko.ai/v1", "name": "Auriko", "doc": "https://docs.auriko.ai", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "grok-4.3": { "id": "grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "minimax-m2-7-highspeed": { "id": "minimax-m2-7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_write": 0.375 } }, "minimax-m2-7": { "id": "minimax-m2-7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_write": 0.375 } }, "qwen-3.6-plus": { "id": "qwen-3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.1, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.5, "output": 2.8 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } } } }, "stepfun-ai": { "id": "stepfun-ai", "env": ["STEPFUN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.stepfun.ai/step_plan/v1", "name": "StepFun AI", "doc": "https://platform.stepfun.ai/docs/en/step-plan/integrations/open-code", "models": { "step-2-16k": { "id": "step-2-16k", "name": "Step 2 (16K)", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-01-01", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 5.21, "output": 16.44, "cache_read": 1.04 } }, "step-tts-2": { "id": "step-tts-2", "name": "Step TTS 2", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "step", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-03-01", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "step-3.5-flash": { "id": "step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "stepaudio-2.5-asr": { "id": "stepaudio-2.5-asr", "name": "StepAudio 2.5 ASR", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "step", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-24", "last_updated": "2026-07-02", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "stepaudio-2.5-tts": { "id": "stepaudio-2.5-tts", "name": "StepAudio 2.5 TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "step", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "step-3.5-flash-2603": { "id": "step-3.5-flash-2603", "name": "Step 3.5 Flash 2603", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "step-3.7-flash": { "id": "step-3.7-flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-06-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 1.15, "cache_read": 0.04 } }, "step-1-32k": { "id": "step-1-32k", "name": "Step 1 (32K)", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-01-01", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 2.05, "output": 9.59, "cache_read": 0.41 } } } }, "vivgrid": { "id": "vivgrid", "env": ["VIVGRID_API_KEY"], "npm": "@ai-sdk/openai", "api": "https://api.vivgrid.com/v1", "name": "Vivgrid", "doc": "https://docs.vivgrid.com/models", "models": { "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.2, "output": 4.2, "cache_read": 0.3 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.1-codex-max": { "id": "gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT 5.6 Luna", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25 } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT 5.6 Terra", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 3.125 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.03 } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT 5.6 Sol", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.28, "output": 0.42 } }, "gemini-3.1-flash-lite-preview": { "id": "gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "cache_write": 1 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "tinfoil": { "id": "tinfoil", "env": ["TINFOIL_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://inference.tinfoil.sh/v1", "name": "Tinfoil", "doc": "https://docs.tinfoil.sh", "models": { "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 262144 }, "cost": { "input": 1.5, "output": 5.25 } }, "llama3-3-70b": { "id": "llama3-3-70b", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 1.75, "output": 2.75 } }, "gpt-oss-safeguard-120b": { "id": "gpt-oss-safeguard-120b", "name": "gpt-oss-safeguard-120b", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-10-29", "last_updated": "2025-10-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6 } }, "nomic-embed-text": { "id": "nomic-embed-text", "name": "Nomic Embed Text v1.5", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2024-02", "last_updated": "2024-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 768 }, "cost": { "input": 0.05, "output": 0 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "gpt-oss-120b", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6 } }, "glm-5-2": { "id": "glm-5-2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 384000, "output": 131072 }, "cost": { "input": 1.5, "output": 5.25 } }, "gemma4-31b": { "id": "gemma4-31b", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.4, "output": 1 } } } }, "mistral": { "id": "mistral", "env": ["MISTRAL_API_KEY"], "npm": "@ai-sdk/mistral", "name": "Mistral", "doc": "https://docs.mistral.ai/getting-started/models/", "models": { "codestral-latest": { "id": "codestral-latest", "name": "Codestral (latest)", "description": "Mistral code model for completions, refactors, and developer IDE workflows", "family": "codestral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-05-29", "last_updated": "2025-01-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 4096 }, "cost": { "input": 0.3, "output": 0.9 } }, "mistral-large-latest": { "id": "mistral-large-latest", "name": "Mistral Large (latest)", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.5, "output": 1.5 } }, "open-mistral-7b": { "id": "open-mistral-7b", "name": "Mistral 7B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2023-09-27", "last_updated": "2023-09-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "output": 8000 }, "cost": { "input": 0.25, "output": 0.25 } }, "devstral-small-2507": { "id": "devstral-small-2507", "name": "Devstral Small", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.3 } }, "ministral-3b-latest": { "id": "ministral-3b-latest", "name": "Ministral 3B (latest)", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.04, "output": 0.04 } }, "pixtral-large-latest": { "id": "pixtral-large-latest", "name": "Pixtral Large (latest)", "description": "Mistral's larger vision model for document-heavy image understanding and chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2024-11-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 2, "output": 6 } }, "mistral-nemo": { "id": "mistral-nemo", "name": "Mistral Nemo", "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.15 } }, "mistral-embed": { "id": "mistral-embed", "name": "Mistral Embed", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "mistral-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-12-11", "last_updated": "2023-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "output": 3072 }, "cost": { "input": 0.1, "output": 0 } }, "mistral-small-2506": { "id": "mistral-small-2506", "name": "Mistral Small 3.2", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.1, "output": 0.3 } }, "ministral-8b-latest": { "id": "ministral-8b-latest", "name": "Ministral 8B (latest)", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.1, "output": 0.1 } }, "open-mixtral-8x22b": { "id": "open-mixtral-8x22b", "name": "Mixtral 8x22B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mixtral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-17", "last_updated": "2024-04-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 64000 }, "cost": { "input": 2, "output": 6 } }, "mistral-medium-latest": { "id": "mistral-medium-latest", "name": "Mistral Medium (latest)", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.5, "output": 7.5 } }, "devstral-small-2505": { "id": "devstral-small-2505", "name": "Devstral Small 2505", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.3 } }, "magistral-small": { "id": "magistral-small", "name": "Magistral Small", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-small", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-03-17", "last_updated": "2025-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.5, "output": 1.5 } }, "mistral-medium-2604": { "id": "mistral-medium-2604", "name": "Mistral Medium 3.5", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.5, "output": 7.5 } }, "mistral-small-latest": { "id": "mistral-small-latest", "name": "Mistral Small (latest)", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.15, "output": 0.6 } }, "open-mixtral-8x7b": { "id": "open-mixtral-8x7b", "name": "Mixtral 8x7B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mixtral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-01", "release_date": "2023-12-11", "last_updated": "2023-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.7, "output": 0.7 } }, "devstral-latest": { "id": "devstral-latest", "name": "Devstral 2", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } }, "mistral-small-2603": { "id": "mistral-small-2603", "name": "Mistral Small 4", "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.15, "output": 0.6 } }, "mistral-medium-2505": { "id": "mistral-medium-2505", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.4, "output": 2 } }, "mistral-large-2411": { "id": "mistral-large-2411", "name": "Mistral Large 2.1", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-18", "last_updated": "2024-11-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 2, "output": 6 } }, "mistral-medium-2508": { "id": "mistral-medium-2508", "name": "Mistral Medium 3.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.4, "output": 2 } }, "open-mistral-nemo": { "id": "open-mistral-nemo", "name": "Open Mistral Nemo", "description": "Legacy model retained for compatibility with older integrations", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.15, "output": 0.15 } }, "magistral-medium-latest": { "id": "magistral-medium-latest", "name": "Magistral Medium (latest)", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-medium", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-03-17", "last_updated": "2025-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2, "output": 5 } }, "devstral-medium-latest": { "id": "devstral-medium-latest", "name": "Devstral 2 (latest)", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } }, "devstral-2512": { "id": "devstral-2512", "name": "Devstral 2", "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } }, "labs-devstral-small-2512": { "id": "labs-devstral-small-2512", "name": "Devstral Small 2", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "status": "deprecated", "cost": { "input": 0, "output": 0 } }, "pixtral-12b": { "id": "pixtral-12b", "name": "Pixtral 12B", "description": "Mistral vision-language model for image understanding and multimodal chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-09-01", "last_updated": "2024-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.15 } }, "mistral-large-2512": { "id": "mistral-large-2512", "name": "Mistral Large 3", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.5, "output": 1.5 } }, "devstral-medium-2507": { "id": "devstral-medium-2507", "name": "Devstral Medium", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } } } }, "cloudflare-workers-ai": { "id": "cloudflare-workers-ai", "env": ["CLOUDFLARE_ACCOUNT_ID", "CLOUDFLARE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1", "name": "Cloudflare Workers AI", "doc": "https://developers.cloudflare.com/workers-ai/models/", "models": { "@cf/ibm-granite/granite-4.0-h-micro": { "id": "@cf/ibm-granite/granite-4.0-h-micro", "name": "Granite 4.0 H Micro", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "granite", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-07", "last_updated": "2025-10-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.017, "output": 0.112 } }, "@cf/moonshotai/kimi-k2.7-code": { "id": "@cf/moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "@cf/moonshotai/kimi-k2.6": { "id": "@cf/moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 256000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "@cf/google/gemma-4-26b-a4b-it": { "id": "@cf/google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.1, "output": 0.3 } }, "@cf/openai/gpt-oss-120b": { "id": "@cf/openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.35, "output": 0.75 } }, "@cf/openai/gpt-oss-20b": { "id": "@cf/openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.2, "output": 0.3 } }, "@cf/mistralai/mistral-small-3.1-24b-instruct": { "id": "@cf/mistralai/mistral-small-3.1-24b-instruct", "name": "Mistral Small 3.1 24B Instruct", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-03-18", "last_updated": "2025-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.351, "output": 0.555 } }, "@cf/nvidia/nemotron-3-120b-a12b": { "id": "@cf/nvidia/nemotron-3-120b-a12b", "name": "Nemotron 3 Super 120B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.5, "output": 1.5 } }, "@cf/zai-org/glm-5.2": { "id": "@cf/zai-org/glm-5.2", "name": "Glm 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "@cf/zai-org/glm-4.7-flash": { "id": "@cf/zai-org/glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.0605, "output": 0.4 } }, "@cf/deepseek-ai/deepseek-r1-distill-qwen-32b": { "id": "@cf/deepseek-ai/deepseek-r1-distill-qwen-32b", "name": "Deepseek R1 Distill Qwen 32B", "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 80000, "output": 80000 }, "cost": { "input": 0.497, "output": 4.881 } }, "@cf/qwen/qwen3-30b-a3b-fp8": { "id": "@cf/qwen/qwen3-30b-a3b-fp8", "name": "Qwen3 30B A3b fp8", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.0509, "output": 0.335 } }, "@cf/qwen/qwen2.5-coder-32b-instruct": { "id": "@cf/qwen/qwen2.5-coder-32b-instruct", "name": "Qwen2.5 Coder 32B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.66, "output": 1 } }, "@cf/qwen/qwq-32b": { "id": "@cf/qwen/qwq-32b", "name": "Qwq 32B", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-03-05", "last_updated": "2025-03-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 24000, "output": 24000 }, "cost": { "input": 0.66, "output": 1 } }, "@cf/meta/llama-3.2-1b-instruct": { "id": "@cf/meta/llama-3.2-1b-instruct", "name": "Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 60000, "output": 60000 }, "cost": { "input": 0.027, "output": 0.201 } }, "@cf/meta/llama-3.2-11b-vision-instruct": { "id": "@cf/meta/llama-3.2-11b-vision-instruct", "name": "Llama 3.2 11B Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.0485, "output": 0.676 } }, "@cf/meta/llama-4-scout-17b-16e-instruct": { "id": "@cf/meta/llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout 17B 16E Instruct", "description": "Open Llama with long-context vision for efficient multimodal agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 16384 }, "cost": { "input": 0.27, "output": 0.85 } }, "@cf/meta/llama-guard-3-8b": { "id": "@cf/meta/llama-guard-3-8b", "name": "Llama Guard 3 8B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-01-22", "last_updated": "2025-01-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.484, "output": 0.03 } }, "@cf/meta/llama-3.3-70b-instruct-fp8-fast": { "id": "@cf/meta/llama-3.3-70b-instruct-fp8-fast", "name": "Llama 3.3 70B Instruct fp8 Fast", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 24000, "output": 24000 }, "cost": { "input": 0.293, "output": 2.253 } }, "@cf/meta/llama-3.1-8b-instruct-fp8": { "id": "@cf/meta/llama-3.1-8b-instruct-fp8", "name": "Llama 3.1 8B Instruct fp8", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2024-07-25", "last_updated": "2024-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.152, "output": 0.287 } }, "@cf/meta/llama-3.2-3b-instruct": { "id": "@cf/meta/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 80000, "output": 80000 }, "cost": { "input": 0.0509, "output": 0.335 } }, "@cf/aisingapore/gemma-sea-lion-v4-27b-it": { "id": "@cf/aisingapore/gemma-sea-lion-v4-27b-it", "name": "Gemma Sea Lion V4 27B It", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.351, "output": 0.555 } } } }, "bailing": { "id": "bailing", "env": ["BAILING_API_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.tbox.cn/api/llm/v1/chat/completions", "name": "Bailing", "doc": "https://alipaytbox.yuque.com/sxs0ba/ling/intro", "models": { "Ring-1T": { "id": "Ring-1T", "name": "Ring-1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "ring", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-06", "release_date": "2025-10", "last_updated": "2025-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0.57, "output": 2.29 } }, "Ling-1T": { "id": "Ling-1T", "name": "Ling-1T", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-10", "last_updated": "2025-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0.57, "output": 2.29 } } } }, "anyapi": { "id": "anyapi", "env": ["ANYAPI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.anyapi.ai/v1", "name": "AnyAPI", "doc": "https://docs.anyapi.ai", "models": { "xai/grok-4.3": { "id": "xai/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 } }, "google/gemini-3-pro-preview": { "id": "google/gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 } }, "openai/o3": { "id": "openai/o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5 }, "provider": { "body": { "service_tier": "priority" } } } } } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "mistralai/devstral-2512": { "id": "mistralai/devstral-2512", "name": "Devstral 2", "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated" }, "mistralai/mistral-large-2512": { "id": "mistralai/mistral-large-2512", "name": "Mistral Large 3", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "anthropic/claude-sonnet-4-5": { "id": "anthropic/claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 } }, "anthropic/claude-opus-4-7": { "id": "anthropic/claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } } }, "anthropic/claude-haiku-4-5": { "id": "anthropic/claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 } }, "anthropic/claude-opus-4-6": { "id": "anthropic/claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } } }, "anthropic/claude-sonnet-4-6": { "id": "anthropic/claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 } }, "cohere/command-r-plus-08-2024": { "id": "cohere/command-r-plus-08-2024", "name": "Command R+", "description": "Cohere's RAG workhorse for long-context enterprise search and tool use", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 } }, "perplexity/sonar-reasoning-pro": { "id": "perplexity/sonar-reasoning-pro", "name": "Sonar Reasoning Pro", "description": "Web-grounded Sonar for multi-step research questions that need cited reasoning", "family": "sonar-reasoning", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 } }, "perplexity/sonar-pro": { "id": "perplexity/sonar-pro", "name": "Sonar Pro", "description": "Deeper Sonar search model with broader retrieval and stronger synthesis", "family": "sonar-pro", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 } }, "deepseek/deepseek-r1": { "id": "deepseek/deepseek-r1", "name": "DeepSeek Reasoner", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 } }, "deepseek/deepseek-chat": { "id": "deepseek/deepseek-chat", "name": "DeepSeek Chat", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 } } } }, "google": { "id": "google", "env": ["GOOGLE_API_KEY", "GOOGLE_GENERATIVE_AI_API_KEY", "GEMINI_API_KEY"], "npm": "@ai-sdk/google", "name": "Google", "doc": "https://ai.google.dev/gemini-api/docs/models", "models": { "gemini-3.1-flash-lite": { "id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "gemini-2.5-flash-preview-tts": { "id": "gemini-2.5-flash-preview-tts", "name": "Gemini 2.5 Flash Preview TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gemini-flash", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-05-01", "last_updated": "2025-05-01", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 8192, "output": 16384 }, "cost": { "input": 0.5, "output": 10 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "gemma-4-31b-it": { "id": "gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 } }, "gemini-2.0-flash": { "id": "gemini-2.0-flash", "name": "Gemini 2.0 Flash", "description": "Earlier Gemini Flash workhorse for responsive multimodal apps and tool use", "family": "gemini-flash", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "gemini-embedding-001": { "id": "gemini-embedding-001", "name": "Gemini Embedding 001", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "gemini", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-05", "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2048, "output": 1 }, "cost": { "input": 0.15, "output": 0 } }, "gemini-3.1-pro-preview-customtools": { "id": "gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "gemini-flash-lite-latest": { "id": "gemini-flash-lite-latest", "name": "Gemini Flash-Lite Latest", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "gemini-3-pro-image-preview": { "id": "gemini-3-pro-image-preview", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 2, "output": 120 } }, "gemini-2.5-flash-image": { "id": "gemini-2.5-flash-image", "name": "Nano Banana", "description": "Nano Banana image model for fast generation, edits, and character-consistent assets", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.3, "output": 30, "cache_read": 0.075 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01, "input_audio": 0.3 } }, "gemini-omni-flash-preview": { "id": "gemini-omni-flash-preview", "name": "Gemini Omni Flash Preview", "description": "Video generation and editing model for fast, conversational text- and image-to-video workflows", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "video"], "output": ["video"] }, "open_weights": false, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 1.5, "output": 17.5 } }, "gemini-3.1-flash-image-preview": { "id": "gemini-3.1-flash-image-preview", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.5, "output": 60 } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "gemma-4-26b-a4b-it": { "id": "gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 } }, "gemini-3-pro-preview": { "id": "gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "status": "deprecated", "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1 } }, "gemini-2.5-pro-preview-tts": { "id": "gemini-2.5-pro-preview-tts", "name": "Gemini 2.5 Pro Preview TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gemini-flash", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-05-01", "last_updated": "2025-05-01", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 8192, "output": 16384 }, "cost": { "input": 1, "output": 20 } }, "gemini-flash-latest": { "id": "gemini-flash-latest", "name": "Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075, "input_audio": 1 } }, "gemini-3.1-flash-lite-preview": { "id": "gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Legacy model retained for compatibility with older integrations", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "status": "deprecated", "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "gemini-2.0-flash-lite": { "id": "gemini-2.0-flash-lite", "name": "Gemini 2.0 Flash-Lite", "description": "Legacy model retained for compatibility with older integrations", "family": "gemini-flash-lite", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.075, "output": 0.3 } } } }, "opencode-go": { "id": "opencode-go", "env": ["OPENCODE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://opencode.ai/zen/go/v1", "name": "OpenCode Go", "doc": "https://opencode.ai/docs/zen", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax-M2.5", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax-m2.5", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 65536 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "qwen3.7-plus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.04, "cache_write": 0.5, "tiers": [ { "input": 1.2, "output": 4.8, "cache_read": 0.12, "cache_write": 1.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1.2, "output": 4.8, "cache_read": 0.12, "cache_write": 1.5 } } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "qwen3.7-max", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.5, "cache_write": 3.125 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 32768 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.0145 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "minimax-m3": { "id": "minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax-m3", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-31", "last_updated": "2026-05-31", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "tiers": [ { "input": 0.6, "output": 2.4, "cache_read": 0.12, "tier": { "type": "context", "size": 512000 } } ], "context_over_200k": { "input": 0.6, "output": 2.4, "cache_read": 0.12 } } }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen3.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.2, "output": 1.2, "cache_read": 0.02, "cache_write": 0.25 } }, "minimax-m2.7": { "id": "minimax-m2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax-m2.7", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-10", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "status": "deprecated", "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "mimo-v2.5": { "id": "mimo-v2.5", "name": "MiMo V2.5", "description": "MiMo omni model for text, image, video, audio, and agents", "family": "mimo-v2.5", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "mimo-v2-omni": { "id": "mimo-v2-omni", "name": "MiMo V2 Omni", "description": "Legacy model retained for compatibility with older integrations", "family": "mimo-v2-omni", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text", "image", "audio", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2, "cache_read": 0.08 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-10", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "mimo-v2-pro": { "id": "mimo-v2-pro", "name": "MiMo V2 Pro", "description": "Legacy model retained for compatibility with older integrations", "family": "mimo-v2-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 128000 }, "status": "deprecated", "cost": { "input": 1, "output": 3, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "mimo-v2.5-pro": { "id": "mimo-v2.5-pro", "name": "MiMo V2.5 Pro", "description": "MiMo pro model for strong multimodal reasoning and agent execution", "family": "mimo-v2.5-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 128000 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.0145 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "Legacy model retained for compatibility with older integrations", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 32768 }, "status": "deprecated", "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } } } }, "digitalocean": { "id": "digitalocean", "env": ["DIGITALOCEAN_ACCESS_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://inference.do-ai.run/v1", "name": "DigitalOcean", "doc": "https://docs.digitalocean.com/products/gradient-ai-platform/details/models/", "models": { "anthropic-claude-haiku-4.5": { "id": "anthropic-claude-haiku-4.5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 1, "cache_write": 1.25 } }, "openai-gpt-image-1": { "id": "openai-gpt-image-1", "name": "GPT Image 1", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-04-24", "last_updated": "2025-04-24", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 40, "cache_read": 1.25 } }, "e5-large-v2": { "id": "e5-large-v2", "name": "E5 Large v2", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-05-19", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 1024 }, "cost": { "input": 0.02, "output": 0 } }, "bge-m3": { "id": "bge-m3", "name": "BGE M3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-01-30", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 1024 }, "cost": { "input": 0.02, "output": 0 } }, "mistral-3-14B": { "id": "mistral-3-14B", "name": "Ministral 3 14B Instruct", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-15", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 128000 }, "cost": { "input": 0.2, "output": 0.2 } }, "nemotron-3-ultra-550b": { "id": "nemotron-3-ultra-550b", "name": "Nemotron 3 Ultra", "description": "Flagship Nemotron model for high-throughput reasoning and complex agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax-m2.5", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2026-02-12", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128000 }, "status": "beta", "cost": { "input": 0.3, "output": 1.2 } }, "openai-gpt-5.4-nano": { "id": "openai-gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "deepseek-v3": { "id": "deepseek-v3", "name": "DeepSeek V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-12-26", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 131072 } }, "openai-gpt-image-2": { "id": "openai-gpt-image-2", "name": "GPT Image 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-04-24", "last_updated": "2025-04-24", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "openai-gpt-5.2": { "id": "openai-gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "deepseek-r1-distill-llama-70b": { "id": "deepseek-r1-distill-llama-70b", "name": "DeepSeek R1 Distill Llama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-01-30", "last_updated": "2025-01-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.99, "output": 0.99 } }, "qwen3-embedding-0.6b": { "id": "qwen3-embedding-0.6b", "name": "Qwen3 Embedding 0.6B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-06-03", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "output": 1024 }, "status": "beta", "cost": { "input": 0.04, "output": 0 } }, "gemma-4-31B-it": { "id": "gemma-4-31B-it", "name": "Gemma 4 31B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.18, "output": 0.5 } }, "llama-4-maverick": { "id": "llama-4-maverick", "name": "Llama 4 Maverick 17B 128E Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.25, "output": 0.87 } }, "anthropic-claude-3.7-sonnet": { "id": "anthropic-claude-3.7-sonnet", "name": "Claude 3.7 Sonnet", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2025-02-24", "last_updated": "2025-02-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "status": "deprecated", "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "openai-gpt-4o-mini": { "id": "openai-gpt-4o-mini", "name": "GPT-4o mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "anthropic-claude-opus-4.7": { "id": "anthropic-claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "deepseek-4-flash": { "id": "deepseek-4-flash", "name": "Deepseek V4 Flash", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-05-27", "last_updated": "2026-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 8192 } }, "anthropic-claude-4.5-haiku": { "id": "anthropic-claude-4.5-haiku", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 1, "cache_write": 1.25 } }, "anthropic-claude-4.6-sonnet": { "id": "anthropic-claude-4.6-sonnet", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.3, "cache_write": 3.75, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.3, "cache_write": 3.75 } } }, "anthropic-claude-sonnet-4": { "id": "anthropic-claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.3, "cache_write": 3.75, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.3, "cache_write": 3.75 } } }, "ministral-3-8b-instruct-2512": { "id": "ministral-3-8b-instruct-2512", "name": "Ministral 3 8B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "anthropic-claude-4.5-sonnet": { "id": "anthropic-claude-4.5-sonnet", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.3, "cache_write": 3.75, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.3, "cache_write": 3.75 } } }, "anthropic-claude-opus-4.6": { "id": "anthropic-claude-opus-4.6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 0.5, "cache_write": 6.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 0.5, "cache_write": 6.25 } } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.975, "output": 4.3, "cache_read": 0.26 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 393216 }, "cost": { "input": 1.74, "output": 3.48 } }, "openai-gpt-oss-20b": { "id": "openai-gpt-oss-20b", "name": "gpt-oss-20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-05", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.45 } }, "qwen-2.5-14b-instruct": { "id": "qwen-2.5-14b-instruct", "name": "Qwen 2.5 14B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 } }, "anthropic-claude-opus-4.8": { "id": "anthropic-claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-05-28", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.05, "output": 4.4, "cache_read": 0.21 } }, "anthropic-claude-opus-4.5": { "id": "anthropic-claude-opus-4.5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic-claude-4.1-opus": { "id": "anthropic-claude-4.1-opus", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "llama3-8b-instruct": { "id": "llama3-8b-instruct", "name": "Llama 3.1 Instruct (8B)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.198, "output": 0.198 } }, "stable-diffusion-3.5-large": { "id": "stable-diffusion-3.5-large", "name": "Stable Diffusion 3.5 Large", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "stable-diffusion", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-10-22", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": true, "limit": { "context": 256, "output": 1 }, "cost": { "input": 0.08, "output": 0 } }, "openai-gpt-5.4-pro": { "id": "openai-gpt-5.4-pro", "name": "GPT-5.4 pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "all-mini-lm-l6-v2": { "id": "all-mini-lm-l6-v2", "name": "All-MiniLM-L6-v2", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2021-08-30", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256, "output": 384 }, "cost": { "input": 0.009, "output": 0 } }, "openai-gpt-image-1.5": { "id": "openai-gpt-image-1.5", "name": "GPT Image 1.5", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 10, "cache_read": 1 } }, "openai-gpt-4o": { "id": "openai-gpt-4o", "name": "GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "anthropic-claude-opus-4": { "id": "anthropic-claude-opus-4", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "bge-reranker-v2-m3": { "id": "bge-reranker-v2-m3", "name": "BGE Reranker v2 M3", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-03-12", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 1 }, "cost": { "input": 0.01, "output": 0 } }, "openai-gpt-5": { "id": "openai-gpt-5", "name": "GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai-o3-mini": { "id": "openai-o3-mini", "name": "o3-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "multi-qa-mpnet-base-dot-v1": { "id": "multi-qa-mpnet-base-dot-v1", "name": "Multi-QA-mpnet-base-dot-v1", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2021-08-30", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 768 }, "cost": { "input": 0.009, "output": 0 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.5, "output": 2.7 } }, "llama3.3-70b-instruct": { "id": "llama3.3-70b-instruct", "name": "Llama 3.3 Instruct 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.65, "output": 0.65 } }, "gte-large-en-v1.5": { "id": "gte-large-en-v1.5", "name": "GTE Large (v1.5)", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-03-27", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 1024 }, "cost": { "input": 0.09, "output": 0 } }, "openai-gpt-5.4-mini": { "id": "openai-gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai-gpt-5.5": { "id": "openai-gpt-5.5", "name": "GPT-5.5", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "nemotron-3-nano-30b": { "id": "nemotron-3-nano-30b", "name": "Nemotron 3 Nano 30B A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "openai-gpt-oss-120b": { "id": "openai-gpt-oss-120b", "name": "gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-05", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.1, "output": 0.7 } }, "openai-gpt-5-nano": { "id": "openai-gpt-5-nano", "name": "GPT-5 nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "wan2-2-t2v-a14b": { "id": "wan2-2-t2v-a14b", "name": "Wan2.2-T2V-A14B", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-07-28", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": true, "limit": { "context": 100, "output": 1 }, "cost": { "input": 0.6, "output": 0 } }, "mistral-7b-instruct-v0.3": { "id": "mistral-7b-instruct-v0.3", "name": "Mistral 7B Instruct v0.3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-05-22", "last_updated": "2024-05-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 } }, "qwen3-tts-voicedesign": { "id": "qwen3-tts-voicedesign", "name": "Qwen3 TTS VoiceDesign", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 32768, "output": 1 } }, "mistral-nemo-instruct-2407": { "id": "mistral-nemo-instruct-2407", "name": "Mistral Nemo Instruct", "description": "Legacy model retained for compatibility with older integrations", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "status": "deprecated", "cost": { "input": 0.3, "output": 0.3 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4 } }, "deepseek-3.2": { "id": "deepseek-3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-02", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.5, "output": 1.6 } }, "openai-o1": { "id": "openai-o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen 3.5 397B A17B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen3.5", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 81920 }, "cost": { "input": 0.55, "output": 3.5 } }, "qwen3-coder-flash": { "id": "qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.45, "output": 1.7 } }, "openai-gpt-5.3-codex": { "id": "openai-gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai-gpt-4.1": { "id": "openai-gpt-4.1", "name": "GPT-4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "anthropic-claude-3.5-haiku": { "id": "anthropic-claude-3.5-haiku", "name": "Claude 3.5 Haiku", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-haiku", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-11-05", "last_updated": "2024-11-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "openai-gpt-5.4": { "id": "openai-gpt-5.4", "name": "GPT-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "arcee-trinity-large-thinking": { "id": "arcee-trinity-large-thinking", "name": "Trinity Large Thinking", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "trinity", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 128000 }, "status": "beta", "cost": { "input": 0.25, "output": 0.9, "cache_read": 0.06 } }, "openai-o3": { "id": "openai-o3", "name": "o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "anthropic-claude-3.5-sonnet": { "id": "anthropic-claude-3.5-sonnet", "name": "Claude 3.5 Sonnet", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-06-20", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "status": "deprecated", "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "openai-gpt-5-mini": { "id": "openai-gpt-5-mini", "name": "GPT-5 mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "nvidia-nemotron-3-super-120b": { "id": "nvidia-nemotron-3-super-120b", "name": "Nemotron-3-Super-120B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02", "release_date": "2026-03-11", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "status": "beta", "cost": { "input": 0.3, "output": 0.65 } }, "glm-5": { "id": "glm-5", "name": "GLM 5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "release_date": "2026-02-11", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 128000 }, "cost": { "input": 1, "output": 3.2 } }, "openai-gpt-5.2-pro": { "id": "openai-gpt-5.2-pro", "name": "GPT-5.2 pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "anthropic-claude-3-opus": { "id": "anthropic-claude-3-opus", "name": "Claude 3 Opus", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-opus", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08", "release_date": "2024-02-29", "last_updated": "2024-02-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "status": "deprecated", "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic-claude-fable-5": { "id": "anthropic-claude-fable-5", "name": "Anthropic Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-06-09", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "nemotron-3-nano-omni": { "id": "nemotron-3-nano-omni", "name": "Nemotron Nano 3 Omni", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.5, "output": 0.9 } }, "alibaba-qwen3-32b": { "id": "alibaba-qwen3-32b", "name": "Qwen3-32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-30", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 40960 }, "cost": { "input": 0.25, "output": 0.55 } }, "nemotron-nano-12b-v2-vl": { "id": "nemotron-nano-12b-v2-vl", "name": "Nemotron Nano 12B v2 VL", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-01", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.2, "output": 0.6 } }, "openai-gpt-5.1-codex-max": { "id": "openai-gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "fal-ai/fast-sdxl": { "id": "fal-ai/fast-sdxl", "name": "Fast SDXL", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "stable-diffusion", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-07-26", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": true, "limit": { "context": 0, "output": 0 } }, "fal-ai/elevenlabs/tts/multilingual-v2": { "id": "fal-ai/elevenlabs/tts/multilingual-v2", "name": "ElevenLabs Multilingual TTS v2", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "elevenlabs", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-08-22", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "fal-ai/flux/schnell": { "id": "fal-ai/flux/schnell", "name": "FLUX.1 [schnell]", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-08-01", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": true, "limit": { "context": 0, "output": 0 } }, "fal-ai/stable-audio-25/text-to-audio": { "id": "fal-ai/stable-audio-25/text-to-audio", "name": "Stable Audio 2.5 (Text-to-Audio)", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-10-08", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } } } }, "subconscious": { "id": "subconscious", "env": ["SUBCONSCIOUS_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://api.subconscious.dev/v1", "name": "Subconscious", "doc": "https://docs.subconscious.dev", "models": { "subconscious/glm-5.2": { "id": "subconscious/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "subconscious/tim-qwen3.6-27b": { "id": "subconscious/tim-qwen3.6-27b", "name": "TIM-Qwen3.6 27B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 5000 }, "cost": { "input": 0.3, "output": 3, "cache_read": 0.15 } } } }, "venice": { "id": "venice", "env": ["VENICE_API_KEY"], "npm": "venice-ai-sdk-provider", "name": "Venice AI", "doc": "https://docs.venice.ai", "models": { "z-ai-glm-5-turbo": { "id": "z-ai-glm-5-turbo", "name": "GLM 5 Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32768 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "grok-4-20-multi-agent": { "id": "grok-4-20-multi-agent", "name": "Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "release_date": "2026-03-12", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 128000 }, "cost": { "input": 1.42, "output": 2.83, "cache_read": 0.23, "tiers": [ { "input": 2.83, "output": 5.67, "cache_read": 0.45, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.83, "output": 5.67, "cache_read": 0.45 } } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.138, "output": 0.275, "cache_read": 0.028 } }, "google-gemma-4-31b-it": { "id": "google-gemma-4-31b-it", "name": "Google Gemma 4 31B Instruct", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-03", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.12, "output": 0.36, "cache_read": 0.09 } }, "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-20", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.75, "output": 3.5, "cache_read": 0.16 } }, "openai-gpt-56-terra-pro": { "id": "openai-gpt-56-terra-pro", "name": "GPT-5.6 Terra Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3.125, "output": 18.75, "cache_read": 0.3125, "cache_write": 3.90625 } }, "qwen3-235b-a22b-instruct-2507": { "id": "qwen3-235b-a22b-instruct-2507", "name": "Qwen 3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-29", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.75 } }, "openai-gpt-56-sol-pro": { "id": "openai-gpt-56-sol-pro", "name": "GPT-5.6 Sol Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 6.25, "output": 37.5, "cache_read": 0.625, "cache_write": 7.8125 } }, "nvidia-nemotron-cascade-2-30b-a3b": { "id": "nvidia-nemotron-cascade-2-30b-a3b", "name": "Nemotron Cascade 2 30B A3B", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-24", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.14, "output": 0.8 } }, "claude-opus-4-7-fast": { "id": "claude-opus-4-7-fast", "name": "Claude Opus 4.7 Fast", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-05-14", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 36, "output": 180, "cache_read": 3.6, "cache_write": 45 } }, "openai-gpt-55-pro": { "id": "openai-gpt-55-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-24", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 }, "cost": { "input": 37.5, "output": 225 } }, "llama-3.3-70b": { "id": "llama-3.3-70b", "name": "Llama 3.3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "release_date": "2025-04-06", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.7, "output": 2.8 } }, "qwen3-5-397b-a17b": { "id": "qwen3-5-397b-a17b", "name": "Qwen 3.5 397B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-16", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.75, "output": 4.5 } }, "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-12-06", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 198000, "output": 32768 }, "cost": { "input": 6, "output": 30, "cache_read": 0.6, "cache_write": 7.5 } }, "zai-org-glm-5": { "id": "zai-org-glm-5", "name": "GLM 5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 32000 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "qwen3-5-35b-a3b": { "id": "qwen3-5-35b-a3b", "name": "Qwen 3.5 35B A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-25", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.3125, "output": 1.25, "cache_read": 0.15625 } }, "venice-uncensored-role-play": { "id": "venice-uncensored-role-play", "name": "Venice Role Play Uncensored", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "venice", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-02-20", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.5, "output": 2 } }, "qwen-3-6-plus": { "id": "qwen-3-6-plus", "name": "Qwen 3.6 Plus Uncensored", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-06", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.625, "output": 3.75, "cache_read": 0.0625, "cache_write": 0.78, "tiers": [ { "input": 2.5, "output": 7.5, "cache_read": 0.0625, "cache_write": 0.78, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2.5, "output": 7.5, "cache_read": 0.0625, "cache_write": 0.78 } } }, "zai-org-glm-4.6": { "id": "zai-org-glm-4.6", "name": "GLM 4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2024-04-01", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 16384 }, "cost": { "input": 0.43, "output": 1.75, "cache_read": 0.08 } }, "openai-gpt-4o-2024-11-20": { "id": "openai-gpt-4o-2024-11-20", "name": "GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2026-02-28", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 3.125, "output": 12.5 } }, "grok-4-3": { "id": "grok-4-3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-18", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 1.42, "output": 2.83, "cache_read": 0.23, "tiers": [ { "input": 2.83, "output": 5.67, "cache_read": 0.45, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.83, "output": 5.67, "cache_read": 0.45 } } }, "qwen-3-7-plus": { "id": "qwen-3-7-plus", "name": "Qwen 3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 2, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 1.5, "output": 6, "cache_read": 0.15, "cache_write": 1.875, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1.5, "output": 6, "cache_read": 0.15, "cache_write": 1.875 } } }, "minimax-m27": { "id": "minimax-m27", "name": "MiniMax M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 32768 }, "cost": { "input": 0.375, "output": 1.5, "cache_read": 0.06875 } }, "kimi-k2-7-code": { "id": "kimi-k2-7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-13", "last_updated": "2026-06-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.75, "output": 3.5, "cache_read": 0.16 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 1.65, "output": 3.301, "cache_read": 0.33 } }, "qwen3-235b-a22b-thinking-2507": { "id": "qwen3-235b-a22b-thinking-2507", "name": "Qwen 3 235B A22B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "release_date": "2025-04-29", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.45, "output": 3.5 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-01-15", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 198000, "output": 64000 }, "cost": { "input": 3.75, "output": 18.75, "cache_read": 0.375, "cache_write": 4.69 } }, "openai-gpt-4o-mini-2024-07-18": { "id": "openai-gpt-4o-mini-2024-07-18", "name": "GPT-4o Mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2026-02-28", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.1875, "output": 0.75, "cache_read": 0.09375 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 6, "output": 30, "cache_read": 0.6, "cache_write": 7.5 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-29", "last_updated": "2026-07-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "openai-gpt-53-codex": { "id": "openai-gpt-53-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-24", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 2.19, "output": 17.5, "cache_read": 0.219 } }, "openai-gpt-54": { "id": "openai-gpt-54", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 131072 }, "cost": { "input": 3.13, "output": 18.8, "cache_read": 0.313 } }, "openai-gpt-56-terra": { "id": "openai-gpt-56-terra", "name": "GPT-5.6 Terra", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3.125, "output": 18.75, "cache_read": 0.3125, "cache_write": 3.90625 } }, "google-gemma-3-27b-it": { "id": "google-gemma-3-27b-it", "name": "Google Gemma 3 27B Instruct", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-11-04", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 16384 }, "cost": { "input": 0.12, "output": 0.2 } }, "openai-gpt-56-luna": { "id": "openai-gpt-56-luna", "name": "GPT-5.6 Luna", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.25, "output": 7.5, "cache_read": 0.125, "cache_write": 1.5625 } }, "aion-labs-aion-3-0-mini": { "id": "aion-labs-aion-3-0-mini", "name": "Aion 3.0 Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.875, "output": 1.75, "cache_read": 0.225 } }, "claude-opus-4-8-fast": { "id": "claude-opus-4-8-fast", "name": "Claude Opus 4.8 Fast", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 12, "output": 60, "cache_read": 1.2, "cache_write": 15 } }, "gemma-4-uncensored": { "id": "gemma-4-uncensored", "name": "Gemma 4 Uncensored", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-04-13", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.1625, "output": 0.5 } }, "qwen3-next-80b": { "id": "qwen3-next-80b", "name": "Qwen 3 Next 80b", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-29", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.35, "output": 1.9 } }, "mistral-small-2603": { "id": "mistral-small-2603", "name": "Mistral Small 4", "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.1875, "output": 0.75 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 6, "output": 30, "cache_read": 0.6, "cache_write": 7.5 } }, "zai-org-glm-4.7-flash": { "id": "zai-org-glm-4.7-flash", "name": "GLM 4.7 Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-29", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.125, "output": 0.5 } }, "google-gemma-4-26b-a4b-it": { "id": "google-gemma-4-26b-a4b-it", "name": "Google Gemma 4 26B A4B Instruct", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.13, "output": 0.4, "cache_read": 0.05 } }, "grok-4-5": { "id": "grok-4-5", "name": "Grok 4.5", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-07", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 32000 }, "cost": { "input": 2.27, "output": 6.8, "cache_read": 0.57, "tiers": [ { "input": 4.53, "output": 13.6, "cache_read": 1.13, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4.53, "output": 13.6, "cache_read": 1.13 } } }, "grok-4-20": { "id": "grok-4-20", "name": "Grok 4.20", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-03-12", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 128000 }, "cost": { "input": 1.42, "output": 2.83, "cache_read": 0.23, "tiers": [ { "input": 2.83, "output": 5.67, "cache_read": 0.45, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.83, "output": 5.67, "cache_read": 0.45 } } }, "hermes-3-llama-3.1-405b": { "id": "hermes-3-llama-3.1-405b", "name": "Hermes 3 Llama 3.1 405b", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "hermes", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2025-09-25", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.1, "output": 3 } }, "aion-labs-aion-2-0": { "id": "aion-labs-aion-2-0", "name": "Aion 2.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "release_date": "2026-03-24", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 1, "output": 2, "cache_read": 0.25 } }, "qwen3-vl-235b-a22b": { "id": "qwen3-vl-235b-a22b", "name": "Qwen3 VL 235B", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-01-16", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.21, "output": 1.9, "cache_read": 0.1 } }, "xiaomi-mimo-v2-5": { "id": "xiaomi-mimo-v2-5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-06-11", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.05 } }, "venice-uncensored-1-2": { "id": "venice-uncensored-1-2", "name": "Venice Uncensored 1.2", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "venice", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-04-01", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.2, "output": 0.9 } }, "gemini-3-5-flash": { "id": "gemini-3-5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-22", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.55, "output": 9.45, "cache_read": 0.155, "cache_write": 0.086 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-10", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 12, "output": 60, "cache_read": 1.2, "cache_write": 15 } }, "openai-gpt-oss-120b": { "id": "openai-gpt-oss-120b", "name": "OpenAI GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-11-06", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.07, "output": 0.3 } }, "zai-org-glm-5-2": { "id": "zai-org-glm-5-2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-16", "last_updated": "2026-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "nvidia-nemotron-3-nano-30b-a3b": { "id": "nvidia-nemotron-3-nano-30b-a3b", "name": "NVIDIA Nemotron 3 Nano 30B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.075, "output": 0.3 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 6, "output": 30, "cache_read": 0.6, "cache_write": 7.5 } }, "mistral-small-3-2-24b-instruct": { "id": "mistral-small-3-2-24b-instruct", "name": "Mistral Small 3.2 24B Instruct", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-01-15", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.09375, "output": 0.25 } }, "openai-gpt-54-pro": { "id": "openai-gpt-54-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 }, "cost": { "input": 37.5, "output": 225, "tiers": [ { "input": 75, "output": 337.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 75, "output": 337.5 } } }, "minimax-m3-preview": { "id": "minimax-m3-preview", "name": "MiniMax M3 Preview", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax-m3", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "release_date": "2026-06-12", "last_updated": "2026-06-13", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 65536 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "gemini-3-1-pro-preview": { "id": "gemini-3-1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.5, "cache_write": 0.5, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 0.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 0.5 } } }, "qwen3-coder-480b-a35b-instruct-turbo": { "id": "qwen3-coder-480b-a35b-instruct-turbo", "name": "Qwen 3 Coder 480B Turbo", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-01-27", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.35, "output": 1.5, "cache_read": 0.04 } }, "qwen3-6-27b": { "id": "qwen3-6-27b", "name": "Qwen 3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-24", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.325, "output": 3.25 } }, "openai-gpt-52-codex": { "id": "openai-gpt-52-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08", "release_date": "2025-01-15", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 272000, "output": 65536 }, "cost": { "input": 2.19, "output": 17.5, "cache_read": 0.219 } }, "qwen-3-7-max": { "id": "qwen-3-7-max", "name": "Qwen 3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-05-22", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.7, "output": 8.05, "cache_read": 0.27, "cache_write": 3.35 } }, "olafangensan-glm-4.7-flash-heretic": { "id": "olafangensan-glm-4.7-flash-heretic", "name": "GLM 4.7 Flash Heretic", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-04", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 24000 }, "cost": { "input": 0.07, "output": 0.4, "cache_read": 0.035 } }, "openai-gpt-54-mini": { "id": "openai-gpt-54-mini", "name": "GPT-5.4 Mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-27", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.9375, "output": 5.625, "cache_read": 0.09375 } }, "grok-build-0-1": { "id": "grok-build-0-1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 4, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2, "output": 4, "cache_read": 0.4 } } }, "zai-org-glm-4.7": { "id": "zai-org-glm-4.7", "name": "GLM 4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-24", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 16384 }, "cost": { "input": 0.55, "output": 2.65, "cache_read": 0.11 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3.6, "output": 18, "cache_read": 0.36, "cache_write": 4.5 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-19", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.7, "output": 3.75, "cache_read": 0.07 } }, "openai-gpt-55": { "id": "openai-gpt-55", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 131072 }, "cost": { "input": 6.25, "output": 37.5, "cache_read": 0.625, "tiers": [ { "input": 12.5, "output": 56.25, "cache_read": 1.25, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 12.5, "output": 56.25, "cache_read": 1.25 } } }, "qwen3-5-9b": { "id": "qwen3-5-9b", "name": "Qwen 3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-05", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.15 } }, "mercury-2": { "id": "mercury-2", "name": "Mercury 2", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "mercury", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-20", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 50000 }, "cost": { "input": 0.3125, "output": 0.9375, "cache_read": 0.03125 } }, "zai-org-glm-5-1": { "id": "zai-org-glm-5-1", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 80000 }, "cost": { "input": 1.54, "output": 4.84, "cache_read": 0.286 } }, "openai-gpt-56-luna-pro": { "id": "openai-gpt-56-luna-pro", "name": "GPT-5.6 Luna Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.25, "output": 7.5, "cache_read": 0.125, "cache_write": 1.5625 } }, "kimi-k2-5": { "id": "kimi-k2-5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2024-04", "release_date": "2026-01-27", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.56, "output": 3.5, "cache_read": 0.22 } }, "z-ai-glm-5v-turbo": { "id": "z-ai-glm-5v-turbo", "name": "GLM 5V Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-06-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32768 }, "cost": { "input": 1.5, "output": 5, "cache_read": 0.3 } }, "nvidia-nemotron-3-ultra-550b-a55b": { "id": "nvidia-nemotron-3-ultra-550b-a55b", "name": "NVIDIA Nemotron 3 Ultra", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.625, "output": 3.125, "cache_read": 0.1875 } }, "aion-labs-aion-3-0": { "id": "aion-labs-aion-3-0", "name": "Aion 3.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 3.75, "output": 7.5, "cache_read": 0.9375 } }, "openai-gpt-52": { "id": "openai-gpt-52", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-13", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 272000, "output": 65536 }, "cost": { "input": 2.19, "output": 17.5, "cache_read": 0.219 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-12-04", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 160000, "output": 32768 }, "cost": { "input": 0.33, "output": 0.48, "cache_read": 0.16 } }, "openai-gpt-56-sol": { "id": "openai-gpt-56-sol", "name": "GPT-5.6 Sol", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 6.25, "output": 37.5, "cache_read": 0.625, "cache_write": 7.8125 } }, "minimax-m25": { "id": "minimax-m25", "name": "MiniMax M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 32768 }, "cost": { "input": 0.27, "output": 0.95, "cache_read": 0.03 } }, "llama-3.2-3b": { "id": "llama-3.2-3b", "name": "Llama 3.2 3B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "release_date": "2024-10-03", "last_updated": "2026-06-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.15, "output": 0.6 } } } }, "crossmodel": { "id": "crossmodel", "env": ["CROSSMODEL_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.crossmodel.ai/v1", "name": "CrossModel", "doc": "https://www.crossmodel.ai/docs", "models": { "z-ai/glm-4.7": { "id": "z-ai/glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.47, "output": 2.16, "cache_read": 0.1, "cache_write": 0.47, "tiers": [ { "input": 0.62, "output": 2.47, "cache_read": 0.13, "cache_write": 0.62, "tier": { "type": "context", "size": 32000 } } ] } }, "z-ai/glm-5.1": { "id": "z-ai/glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 1, "output": 3.8, "cache_read": 0.2, "cache_write": 1, "tiers": [ { "input": 1.2, "output": 4.4, "cache_read": 0.3, "cache_write": 1.2, "tier": { "type": "context", "size": 32000 } } ] } }, "z-ai/glm-5.2": { "id": "z-ai/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.2, "output": 4.4, "cache_read": 0.3, "cache_write": 1.2 } }, "z-ai/glm-5": { "id": "z-ai/glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.16, "cache_write": 0.6, "tiers": [ { "input": 0.8, "output": 3.4, "cache_read": 0.2, "cache_write": 0.8, "tier": { "type": "context", "size": 32000 } } ] } }, "z-ai/glm-5-turbo": { "id": "z-ai/glm-5-turbo", "name": "GLM-5-Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.9, "output": 3.7, "cache_read": 0.18, "cache_write": 0.9, "tiers": [ { "input": 1.1, "output": 4.3, "cache_read": 0.27, "cache_write": 1.1, "tier": { "type": "context", "size": 32000 } } ] } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02, "cache_write": 0.2 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 2.5, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 5 } } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075, "cache_write": 0.75 } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075, "cache_write": 0.15 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "cache_write": 10, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1, "cache_write": 10 } } }, "xiaomi/mimo-v2.5": { "id": "xiaomi/mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.16, "output": 0.32, "cache_read": 0.004, "cache_write": 0.16 } }, "xiaomi/mimo-v2.5-pro": { "id": "xiaomi/mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.47, "output": 0.94, "cache_read": 0.005, "cache_write": 0.47 } }, "anthropic/claude-opus-4-7": { "id": "anthropic/claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-5": { "id": "anthropic/claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "anthropic/claude-opus-4-8": { "id": "anthropic/claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-fable-5": { "id": "anthropic/claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "anthropic/claude-haiku-4-5": { "id": "anthropic/claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-sonnet-4-6": { "id": "anthropic/claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Hy3 preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.19, "output": 0.63, "cache_read": 0.063, "cache_write": 0.19, "tiers": [ { "input": 0.25, "output": 1, "cache_read": 0.094, "cache_write": 0.25, "tier": { "type": "context", "size": 16000 } }, { "input": 0.32, "output": 1.25, "cache_read": 0.125, "cache_write": 0.32, "tier": { "type": "context", "size": 32000 } } ] } }, "moonshot/kimi-k2.7-code": { "id": "moonshot/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 1, "output": 4.16, "cache_read": 0.18, "cache_write": 1 } }, "moonshot/kimi-k2.5": { "id": "moonshot/kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.62, "output": 3.3, "cache_read": 0.11, "cache_write": 0.62 } }, "moonshot/kimi-k2.6": { "id": "moonshot/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 1, "output": 4.16, "cache_read": 0.18, "cache_write": 1 } }, "gemini/gemini-2.5-pro": { "id": "gemini/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "cache_write": 1.25, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 2.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 2.5 } } }, "gemini/gemini-2.5-flash": { "id": "gemini/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "cache_write": 0.3 } }, "gemini/gemini-3.5-flash": { "id": "gemini/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "cache_write": 1.5 } }, "gemini/gemini-2.5-flash-lite": { "id": "gemini/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01, "cache_write": 0.1 } }, "gemini/gemini-3.1-pro-preview": { "id": "gemini/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "cache_write": 2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "cache_write": 4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4, "cache_write": 4 } } }, "gemini/gemini-3-flash-preview": { "id": "gemini/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.5 } }, "qwen/qwen3.7-plus": { "id": "qwen/qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.32, "output": 1.25, "cache_read": 0.032, "cache_write": 0.4, "tiers": [ { "input": 0.96, "output": 3.75, "cache_read": 0.096, "cache_write": 1.2, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.96, "output": 3.75, "cache_read": 0.096, "cache_write": 1.2 } } }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.88, "output": 5.63, "cache_read": 0.375, "cache_write": 2.35 } }, "qwen/qwen3.6-flash": { "id": "qwen/qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.19, "output": 1.13, "cache_read": 0.019, "cache_write": 0.24, "tiers": [ { "input": 0.75, "output": 4.5, "cache_read": 0.075, "cache_write": 0.94, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.75, "output": 4.5, "cache_read": 0.075, "cache_write": 0.94 } } }, "qwen/qwen3.6-plus": { "id": "qwen/qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.32, "output": 1.88, "cache_read": 0.032, "cache_write": 0.4, "tiers": [ { "input": 1.25, "output": 7.5, "cache_read": 0.124, "cache_write": 1.57, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1.25, "output": 7.5, "cache_read": 0.124, "cache_write": 1.57 } } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.16, "output": 0.32, "cache_read": 0.004, "cache_write": 0.16 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.47, "output": 0.94, "cache_read": 0.005, "cache_write": 0.47 } }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1024000, "output": 512000 }, "cost": { "input": 0.33, "output": 1.32, "cache_read": 0.066, "cache_write": 0.33, "tiers": [ { "input": 0.66, "output": 2.63, "cache_read": 0.132, "cache_write": 0.66, "tier": { "type": "context", "size": 512000 } } ], "context_over_200k": { "input": 0.66, "output": 2.63, "cache_read": 0.132, "cache_write": 0.66 } } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.33, "output": 1.32, "cache_read": 0.066, "cache_write": 0.42 } } } }, "lmstudio": { "id": "lmstudio", "env": ["LMSTUDIO_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "http://127.0.0.1:1234/v1", "name": "LMStudio", "doc": "https://lmstudio.ai/models", "models": { "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3-30b-a3b-2507": { "id": "qwen/qwen3-30b-a3b-2507", "name": "Qwen3 30B A3B 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3-coder-30b": { "id": "qwen/qwen3-coder-30b", "name": "Qwen3 Coder 30B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0 } } } }, "poolside": { "id": "poolside", "env": ["POOLSIDE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://inference.poolside.ai/v1", "name": "Poolside", "doc": "https://platform.poolside.ai", "models": { "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", "name": "Laguna XS.2", "description": "Agentic coding model from Poolside in the XS size class for local deployment", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "poolside/laguna-m.1": { "id": "poolside/laguna-m.1", "name": "Laguna M.1", "description": "Poolside's flagship agentic coding model for long-horizon work", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "poolside/laguna-xs-2.1": { "id": "poolside/laguna-xs-2.1", "name": "Laguna XS 2.1", "description": "Agentic coding model from Poolside in the XS size class for local deployment", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": false, "temperature": true, "release_date": "2026-07-02", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "zenifra": { "id": "zenifra", "env": ["ZENIFRA_AI_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://ai.zenifra.com/v1", "name": "Zenifra", "doc": "https://docs.zenifra.com", "models": { "alibaba/qwen3.6-35b-a3b": { "id": "alibaba/qwen3.6-35b-a3b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "provider": { "shape": "completions" }, "cost": { "input": 0.19, "output": 0.48 } } } }, "zenmux": { "id": "zenmux", "env": ["ZENMUX_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://zenmux.ai/api/v1", "name": "ZenMux", "doc": "https://docs.zenmux.ai", "models": { "inclusionai/ling-1t": { "id": "inclusionai/ling-1t", "name": "Ling-1T", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-10-09", "last_updated": "2025-10-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.56, "output": 2.24, "cache_read": 0.11 } }, "inclusionai/ring-2.6-1t": { "id": "inclusionai/ring-2.6-1t", "name": "inclusionAI: Ring-2.6-1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-12-31", "release_date": "2026-05-07", "last_updated": "2026-05-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 65000 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.06 } }, "inclusionai/ring-1t": { "id": "inclusionai/ring-1t", "name": "Ring-1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-10-12", "last_updated": "2025-10-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.56, "output": 2.24, "cache_read": 0.11 } }, "moonshotai/kimi-k2.7-code-free": { "id": "moonshotai/kimi-k2.7-code-free", "name": "Kimi K2.7 Code (Free)", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "moonshotai/kimi-k2-thinking-turbo": { "id": "moonshotai/kimi-k2-thinking-turbo", "name": "Kimi K2 Thinking Turbo", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 64000 }, "cost": { "input": 1.15, "output": 8, "cache_read": 0.15 } }, "moonshotai/kimi-k2.7-code": { "id": "moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 64000 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2025-01-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 64000 }, "cost": { "input": 0.58, "output": 3.02, "cache_read": 0.1 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2025-01-01", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262140, "output": 262140 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 64000 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "baidu/ernie-5.0-thinking-preview": { "id": "baidu/ernie-5.0-thinking-preview", "name": "ERNIE 5.0", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.84, "output": 3.37 } }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["pdf", "image", "text", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048000, "output": 64000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.31, "cache_write": 4.5 } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["pdf", "image", "text", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048000, "output": 64000 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.07, "cache_write": 1 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["pdf", "image", "text", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048000, "output": 64000 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.03, "cache_write": 1 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-02-19", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "pdf", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048000, "output": 64000 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "cache_write": 4.5 } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "pdf", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048000, "output": 64000 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 1 } }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-03-20", "last_updated": "2025-03-20", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 65530 }, "cost": { "input": 0.25, "output": 1.5 } }, "x-ai/grok-code-fast-1": { "id": "x-ai/grok-code-fast-1", "name": "Grok Code Fast 1", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.2, "output": 1.5, "cache_read": 0.02 } }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "cache_write": 0, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "cache_write": 0, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4, "cache_write": 0 } } }, "x-ai/grok-4.1-fast-non-reasoning": { "id": "x-ai/grok-4.1-fast-non-reasoning", "name": "Grok 4.1 Fast Non Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 64000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "x-ai/grok-4": { "id": "x-ai/grok-4", "name": "Grok 4", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75 } }, "x-ai/grok-4-fast": { "id": "x-ai/grok-4-fast", "name": "Grok 4 Fast", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 64000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "x-ai/grok-4.2-fast-non-reasoning": { "id": "x-ai/grok-4.2-fast-non-reasoning", "name": "Grok 4.2 Fast Non Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 3, "output": 9 } }, "x-ai/grok-4.2-fast": { "id": "x-ai/grok-4.2-fast", "name": "Grok 4.2 Fast", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 3, "output": 9 } }, "x-ai/grok-4.1-fast": { "id": "x-ai/grok-4.1-fast", "name": "Grok 4.1 Fast", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 64000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2 } }, "z-ai/glm-4.7": { "id": "z-ai/glm-4.7", "name": "GLM 4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.28, "output": 1.14, "cache_read": 0.06 } }, "z-ai/glm-4.5": { "id": "z-ai/glm-4.5", "name": "GLM 4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.35, "output": 1.54, "cache_read": 0.07 } }, "z-ai/glm-4.7-flashx": { "id": "z-ai/glm-4.7-flashx", "name": "GLM 4.7 FlashX", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.07, "output": 0.42, "cache_read": 0.01 } }, "z-ai/glm-5.1": { "id": "z-ai/glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-03", "last_updated": "2026-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.8781, "output": 3.5126, "cache_read": 0.1903 } }, "z-ai/glm-4.6": { "id": "z-ai/glm-4.6", "name": "GLM 4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.35, "output": 1.54, "cache_read": 0.07 } }, "z-ai/glm-5.2": { "id": "z-ai/glm-5.2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.5, "cache_read": 0.26 } }, "z-ai/glm-4.6v-flash-free": { "id": "z-ai/glm-4.6v-flash-free", "name": "GLM 4.6V Flash (Free)", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "z-ai/glm-4.7-flash-free": { "id": "z-ai/glm-4.7-flash-free", "name": "GLM 4.7 Flash (Free)", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "z-ai/glm-5.2-free": { "id": "z-ai/glm-5.2-free", "name": "GLM 5.2 (Free)", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "z-ai/glm-4.6v": { "id": "z-ai/glm-4.6v", "name": "GLM 4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.14, "output": 0.42, "cache_read": 0.03 } }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "GLM 5V Turbo", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.726, "output": 3.1946, "cache_read": 0.1743 } }, "z-ai/glm-4.6v-flash": { "id": "z-ai/glm-4.6v-flash", "name": "GLM 4.6V FlashX", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.02, "output": 0.21, "cache_read": 0.0043 } }, "z-ai/glm-4.5-air": { "id": "z-ai/glm-4.5-air", "name": "GLM 4.5 Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.11, "output": 0.56, "cache_read": 0.02 } }, "z-ai/glm-5": { "id": "z-ai/glm-5", "name": "GLM 5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.58, "output": 2.6, "cache_read": 0.14 } }, "z-ai/glm-5-turbo": { "id": "z-ai/glm-5-turbo", "name": "GLM 5 Turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.88, "output": 3.48 } }, "openai/gpt-5.5-instant": { "id": "openai/gpt-5.5-instant", "name": "GPT-5.5 Instant", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12-01", "release_date": "2026-05-05", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "GPT-5.2-Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 21, "output": 168 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12 } }, "openai/gpt-5.1-chat": { "id": "openai/gpt-5.1-chat", "name": "GPT-5.1 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 0.2, "output": 1.25 } }, "openai/gpt-5.3-chat": { "id": "openai/gpt-5.3-chat", "name": "GPT-5.3 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16380 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.75, "output": 14 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT-5.1-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-01-01", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.17 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.75, "output": 14 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "GPT-5.1-Codex-Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.03 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 3.75, "output": 18.75 } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 0.75, "output": 4.5 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 45, "output": 225 } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "GPT-5 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT-5.2-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-01-01", "release_date": "2026-01-15", "last_updated": "2026-01-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.17 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai", "api": "https://zenmux.ai/api/v1" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 12.5, "output": 75, "cache_read": 1.25 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "xiaomi/mimo-v2.5": { "id": "xiaomi/mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.08, "tiers": [ { "input": 0.8, "output": 4, "cache_read": 0.16, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.8, "output": 4, "cache_read": 0.16 } } }, "xiaomi/mimo-v2-omni": { "id": "xiaomi/mimo-v2-omni", "name": "MiMo V2 Omni", "description": "MiMo omni model for text, image, video, audio, and agents", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 265000, "output": 265000 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.08 } }, "xiaomi/mimo-v2-flash": { "id": "xiaomi/mimo-v2-flash", "name": "MiMo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12-01", "release_date": "2025-12-16", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01 } }, "xiaomi/mimo-v2-pro": { "id": "xiaomi/mimo-v2-pro", "name": "MiMo V2 Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 256000 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "xiaomi/mimo-v2.5-pro": { "id": "xiaomi/mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "sapiens-ai/agnes-1.5-pro": { "id": "sapiens-ai/agnes-1.5-pro", "name": "Agnes 1.5 Pro", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-21", "last_updated": "2026-03-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.16, "output": 0.8 } }, "sapiens-ai/agnes-1.5-lite": { "id": "sapiens-ai/agnes-1.5-lite", "name": "Agnes 1.5 Lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-26", "last_updated": "2026-03-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.12, "output": 0.6 } }, "volcengine/doubao-seed-2.0-pro": { "id": "volcengine/doubao-seed-2.0-pro", "name": "Doubao-Seed-2.0-pro", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-02-14", "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.45, "output": 2.24, "cache_read": 0.09, "cache_write": 0.0024 } }, "volcengine/doubao-seed-code": { "id": "volcengine/doubao-seed-code", "name": "Doubao-Seed-Code", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-11", "last_updated": "2025-11-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.17, "output": 1.12, "cache_read": 0.03 } }, "volcengine/doubao-seed-1.8": { "id": "volcengine/doubao-seed-1.8", "name": "Doubao-Seed-1.8", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-18", "last_updated": "2025-12-18", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.11, "output": 0.28, "cache_read": 0.02, "cache_write": 0.0024 } }, "volcengine/doubao-seed-2.0-mini": { "id": "volcengine/doubao-seed-2.0-mini", "name": "Doubao-Seed-2.0-mini", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-02-14", "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.03, "output": 0.28, "cache_read": 0.01, "cache_write": 0.0024 } }, "volcengine/doubao-seed-2.0-lite": { "id": "volcengine/doubao-seed-2.0-lite", "name": "Doubao-Seed-2.0-lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-02-14", "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.09, "output": 0.51, "cache_read": 0.02, "cache_write": 0.0024 } }, "volcengine/doubao-seed-2.0-code": { "id": "volcengine/doubao-seed-2.0-code", "name": "Doubao Seed 2.0 Code", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0.9, "output": 4.48 } }, "anthropic/claude-sonnet-5-free": { "id": "anthropic/claude-sonnet-5-free", "name": "Claude Sonnet 5 (Free)", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "anthropic/claude-3.5-haiku": { "id": "anthropic/claude-3.5-haiku", "name": "Claude 3.5 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2024-11-04", "last_updated": "2024-11-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "anthropic/claude-sonnet-4.5": { "id": "anthropic/claude-sonnet-4.5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-sonnet-5": { "id": "anthropic/claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 4 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-3.7-sonnet": { "id": "anthropic/claude-3.7-sonnet", "name": "Claude 3.7 Sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-02-24", "last_updated": "2025-02-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-fable-5": { "id": "anthropic/claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "anthropic/claude-opus-4.1": { "id": "anthropic/claude-opus-4.1", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.5": { "id": "anthropic/claude-opus-4.5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-18", "last_updated": "2026-02-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-06", "last_updated": "2026-02-06", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Hy3 preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.172, "output": 0.572, "cache_read": 0.058, "cache_write": 0 } }, "stepfun/step-3": { "id": "stepfun/step-3", "name": "Step-3", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "output": 64000 }, "cost": { "input": 0.21, "output": 0.57 } }, "stepfun/step-3.7-flash": { "id": "stepfun/step-3.7-flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 1.15 } }, "stepfun/step-3.7-flash-free": { "id": "stepfun/step-3.7-flash-free", "name": "Step 3.7 Flash (Free)", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0, "output": 0 } }, "stepfun/step-3.5-flash": { "id": "stepfun/step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-02-02", "last_updated": "2026-02-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.1, "output": 0.3 } }, "kuaishou/kat-coder-pro-v2": { "id": "kuaishou/kat-coder-pro-v2", "name": "KAT-Coder-Pro-V2", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-30", "last_updated": "2026-03-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 80000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "qwen/qwen3-coder-plus": { "id": "qwen/qwen3-coder-plus", "name": "Qwen3-Coder-Plus", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "qwen/qwen3.7-plus": { "id": "qwen/qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.08, "cache_write": 0.5, "tiers": [ { "input": 1.2, "output": 4.8, "cache_read": 0.24, "cache_write": 1.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1.2, "output": 4.8, "cache_read": 0.24, "cache_write": 1.5 } } }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.5, "cache_write": 3.125 } }, "qwen/qwen3-max": { "id": "qwen/qwen3-max", "name": "Qwen3-Max-Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-01-23", "last_updated": "2026-01-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 1.2, "output": 6 } }, "qwen/qwen3.5-flash": { "id": "qwen/qwen3.5-flash", "name": "Qwen3.5 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1020000, "output": 1020000 }, "cost": { "input": 0.1, "output": 0.4 } }, "qwen/qwen3.5-plus": { "id": "qwen/qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.8, "output": 4.8 } }, "qwen/qwen3.6-plus": { "id": "qwen/qwen3.6-plus", "name": "Qwen3.6-Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-30", "last_updated": "2026-03-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "deepseek/deepseek-v3.2-exp": { "id": "deepseek/deepseek-v3.2-exp", "name": "DeepSeek-V3.2-Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163000, "output": 64000 }, "cost": { "input": 0.22, "output": 0.33 } }, "deepseek/deepseek-chat": { "id": "deepseek/deepseek-chat", "name": "DeepSeek-V3.2 (Non-thinking Mode)", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.28, "output": 0.42, "cache_read": 0.03 } }, "deepseek/deepseek-v3.2": { "id": "deepseek/deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-05", "last_updated": "2025-12-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.28, "output": 0.43 } }, "minimax/minimax-m2.7-highspeed": { "id": "minimax/minimax-m2.7-highspeed", "name": "MiniMax M2.7 highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131070 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.611, "output": 2.4439 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "minimax/minimax-m2.5-lightning": { "id": "minimax/minimax-m2.5-lightning", "name": "MiniMax M2.5 highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.6, "output": 4.8, "cache_read": 0.06, "cache_write": 0.75 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "MiniMax M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.38 } }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.6, "output": 2.4 } }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.38 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131070 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://zenmux.ai/api/anthropic/v1" }, "cost": { "input": 0.3055, "output": 1.2219 } } } }, "kenari": { "id": "kenari", "env": ["KENARI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://kenari.id/v1", "name": "Kenari", "doc": "https://kenari.id/docs", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0 } }, "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0 } }, "glm-5-1": { "id": "glm-5-1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "gemma-4-31b-it": { "id": "gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "grok-4-3": { "id": "grok-4-3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-7-plus": { "id": "qwen3-7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "kimi-k2-7-code": { "id": "kimi-k2-7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0 } }, "deepseek-v4-flash:free": { "id": "deepseek-v4-flash:free", "name": "DeepSeek V4 Flash (Free)", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "minimax-m3": { "id": "minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "gpt-5-4-mini": { "id": "gpt-5-4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2-5": { "id": "mimo-v2-5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "gpt-5-5": { "id": "gpt-5-5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "glm-5-2": { "id": "glm-5-2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "grok-build-0-1": { "id": "grok-build-0-1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0, "output": 0 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2-5-pro": { "id": "mimo-v2-5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "deepseek-v4-pro:free": { "id": "deepseek-v4-pro:free", "name": "DeepSeek V4 Pro (Free)", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0 } }, "gpt-image-2": { "id": "gpt-image-2", "name": "GPT-Image-2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 272000, "output": 16384 }, "cost": { "input": 0, "output": 0 } } } }, "model-oracle-ai": { "id": "model-oracle-ai", "env": ["MODEL_ORACLE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.modeloracle.com/api/v1", "name": "Model Oracle AI", "doc": "https://modeloracle.com/setup/", "models": { "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "claude-haiku-4.5": { "id": "claude-haiku-4.5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 } }, "o4-mini": { "id": "o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 } }, "claude-opus-4.8": { "id": "claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 } }, "auto": { "id": "auto", "name": "Auto", "description": "Model Oracle AI decision engine that selects and routes among configured coding-agent models", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-29", "last_updated": "2026-07-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 } } } }, "openai": { "id": "openai", "env": ["OPENAI_API_KEY"], "npm": "@ai-sdk/openai", "name": "OpenAI", "doc": "https://platform.openai.com/docs/models", "models": { "o3": { "id": "o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "text-embedding-3-large": { "id": "text-embedding-3-large", "name": "text-embedding-3-large", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-01", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 3072 }, "cost": { "input": 0.13, "output": 0 } }, "gpt-5.2-pro": { "id": "gpt-5.2-pro", "name": "GPT-5.2 Pro", "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "gpt-5.6": { "id": "gpt-5.6", "name": "GPT-5.6", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 10, "output": 60, "cache_read": 1, "cache_write": 12.5 }, "provider": { "body": { "service_tier": "priority" } } }, "pro": { "provider": { "body": { "reasoning": { "mode": "pro" } } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5 } } }, "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-3.5-turbo": { "id": "gpt-3.5-turbo", "name": "GPT-3.5-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2021-09-01", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 0 } }, "gpt-5-pro": { "id": "gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "gpt-4o": { "id": "gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "gpt-4": { "id": "gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 30, "output": 60 } }, "o4-mini": { "id": "o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "o3-pro": { "id": "o3-pro", "name": "o3-pro", "description": "High-effort o3 tier for difficult technical reasoning and careful answers", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 20, "output": 80 } }, "chatgpt-image-latest": { "id": "chatgpt-image-latest", "name": "chatgpt-image-latest", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 0, "input": 0, "output": 0 } }, "gpt-4o-2024-05-13": { "id": "gpt-4o-2024-05-13", "name": "GPT-4o (2024-05-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 5, "output": 15 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "gpt-5-chat-latest": { "id": "gpt-5-chat-latest", "name": "GPT-5 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Codex GPT for repository edits, code review, and practical software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.3-codex-spark": { "id": "gpt-5.3-codex-spark", "name": "GPT-5.3 Codex Spark", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex-spark", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 100000, "output": 32000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.1-codex-max": { "id": "gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.3-chat-latest": { "id": "gpt-5.3-chat-latest", "name": "GPT-5.3 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-4o-2024-08-06": { "id": "gpt-4o-2024-08-06", "name": "GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-08-06", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "text-embedding-ada-002": { "id": "text-embedding-ada-002", "name": "text-embedding-ada-002", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2022-12", "release_date": "2022-12-15", "last_updated": "2022-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 }, "cost": { "input": 0.1, "output": 0 } }, "o3-mini": { "id": "o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "text-embedding-3-small": { "id": "text-embedding-3-small", "name": "text-embedding-3-small", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-01", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 1536 }, "cost": { "input": 0.02, "output": 0 } }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 2, "output": 12, "cache_read": 0.2, "cache_write": 2.5 }, "provider": { "body": { "service_tier": "priority" } } }, "pro": { "provider": { "body": { "reasoning": { "mode": "pro" } } } } } }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25, "tiers": [ { "input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5 } } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gpt-realtime-2.1": { "id": "gpt-realtime-2.1", "name": "GPT-Realtime-2.1", "description": "Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2024-09-30", "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": { "input": ["text", "audio", "image"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 128000, "input": 96000, "output": 32000 }, "cost": { "input": 4, "output": 24, "cache_read": 0.4, "input_audio": 32, "output_audio": 64 } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25 }, "provider": { "body": { "service_tier": "priority" } } }, "pro": { "provider": { "body": { "reasoning": { "mode": "pro" } } } } } }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 3.125, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 6.25, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 6.25 } } }, "gpt-5.1-chat-latest": { "id": "gpt-5.1-chat-latest", "name": "GPT-5.1 Chat", "description": "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.2-chat-latest": { "id": "gpt-5.2-chat-latest", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "o4-mini-deep-research": { "id": "o4-mini-deep-research", "name": "o4-mini-deep-research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-06-26", "last_updated": "2024-06-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "gpt-image-1.5": { "id": "gpt-image-1.5", "name": "gpt-image-1.5", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 0, "input": 0, "output": 0 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "gpt-4o-2024-11-20": { "id": "gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "o1": { "id": "o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "o1-pro": { "id": "o1-pro", "name": "o1-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2025-03-19", "last_updated": "2025-03-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 150, "output": 600 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "o3-deep-research": { "id": "o3-deep-research", "name": "o3-deep-research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-06-26", "last_updated": "2024-06-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 10, "output": 40, "cache_read": 2.5 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gpt-image-1": { "id": "gpt-image-1", "name": "gpt-image-1", "description": "OpenAI image model for production generation, edits, and brand-safe visual workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-04-24", "last_updated": "2025-04-24", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "input": 0, "output": 0 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "gpt-4-turbo": { "id": "gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "gpt-image-1-mini": { "id": "gpt-image-1-mini", "name": "gpt-image-1-mini", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-09-26", "last_updated": "2025-09-26", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 0, "input": 0, "output": 0 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "gpt-5.4-pro": { "id": "gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "gpt-5.5-pro": { "id": "gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "gpt-4o-mini": { "id": "gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 10, "output": 60, "cache_read": 1, "cache_write": 12.5 }, "provider": { "body": { "service_tier": "priority" } } }, "pro": { "provider": { "body": { "reasoning": { "mode": "pro" } } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5 } } }, "gpt-5-codex": { "id": "gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-image-2": { "id": "gpt-image-2", "name": "gpt-image-2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "input": 0, "output": 0 }, "cost": { "input": 5, "output": 30, "cache_read": 1.25 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 12.5, "output": 75, "cache_read": 1.25 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "berget": { "id": "berget", "env": ["BERGET_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.berget.ai/v1", "name": "Berget.AI", "doc": "https://api.berget.ai", "models": { "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-04-27", "last_updated": "2025-04-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.99, "output": 0.99 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.83, "output": 3.85, "cache_read": 0.16 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B Instruct", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["audio", "image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.275, "output": 0.55 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT-OSS-120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.22, "output": 0.83 } }, "mistralai/Mistral-Medium-3.5-128B": { "id": "mistralai/Mistral-Medium-3.5-128B", "name": "Mistral Medium 3.5 128B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-04", "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 1.65, "output": 5.5 } }, "mistralai/Mistral-Small-3.2-24B-Instruct-2506": { "id": "mistralai/Mistral-Small-3.2-24B-Instruct-2506", "name": "Mistral Small 3.2 24B Instruct 2506", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-10-01", "last_updated": "2025-10-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 8192 }, "cost": { "input": 0.33, "output": 0.33 } }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM 4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.77, "output": 2.75 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 32768 }, "cost": { "input": 1.54, "output": 4.84 } } } }, "snowflake-cortex": { "id": "snowflake-cortex", "env": ["SNOWFLAKE_ACCOUNT", "SNOWFLAKE_CORTEX_PAT"], "npm": "@ai-sdk/openai-compatible", "api": "https://${SNOWFLAKE_ACCOUNT}.snowflakecomputing.com/api/v2/cortex/v1", "name": "Snowflake Cortex", "doc": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-rest-api", "models": { "openai-gpt-5.1": { "id": "openai-gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "snowflake-llama3.3-70b": { "id": "snowflake-llama3.3-70b", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 } }, "openai-gpt-5.2": { "id": "openai-gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 16384 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "status": "beta", "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } } }, "deepseek-r1": { "id": "deepseek-r1", "name": "DeepSeek-R1", "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "openai-gpt-5": { "id": "openai-gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "status": "beta" }, "openai-gpt-5.5": { "id": "openai-gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "status": "beta" }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "openai-gpt-5-nano": { "id": "openai-gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "status": "beta" }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 16384 } }, "mistral-large2": { "id": "mistral-large2", "name": "Mistral Large (latest)", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "openai-gpt-4.1": { "id": "openai-gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 16384 } }, "openai-gpt-5.4": { "id": "openai-gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "status": "beta", "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5 }, "provider": { "body": { "service_tier": "priority" } } } } } }, "gemini-3.1-pro": { "id": "gemini-3.1-pro", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 } }, "openai-gpt-5-mini": { "id": "openai-gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "input": 272000, "output": 8192 }, "status": "beta" } } }, "tencent-token-plan": { "id": "tencent-token-plan", "env": ["TENCENT_TOKEN_PLAN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.lkeap.cloud.tencent.com/plan/v3", "name": "Tencent Token Plan", "doc": "https://cloud.tencent.com/document/product/1823/130060", "models": { "hy3": { "id": "hy3", "name": "Hy3", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "github-models": { "id": "github-models", "env": ["GITHUB_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://models.github.ai/inference", "name": "GitHub Models", "doc": "https://docs.github.com/en/github-models", "models": { "ai21-labs/ai21-jamba-1.5-mini": { "id": "ai21-labs/ai21-jamba-1.5-mini", "name": "AI21 Jamba 1.5 Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "jamba", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-08-29", "last_updated": "2024-08-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "ai21-labs/ai21-jamba-1.5-large": { "id": "ai21-labs/ai21-jamba-1.5-large", "name": "AI21 Jamba 1.5 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "jamba", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-08-29", "last_updated": "2024-08-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "core42/jais-30b-chat": { "id": "core42/jais-30b-chat", "name": "JAIS 30b Chat", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "jais", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-03", "release_date": "2023-08-30", "last_updated": "2023-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "xai/grok-3-mini": { "id": "xai/grok-3-mini", "name": "Grok 3 Mini", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-09", "last_updated": "2024-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "xai/grok-3": { "id": "xai/grok-3", "name": "Grok 3", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-09", "last_updated": "2024-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3.5-moe-instruct": { "id": "microsoft/phi-3.5-moe-instruct", "name": "Phi-3.5-MoE instruct (128k)", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-08-20", "last_updated": "2024-08-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3-small-128k-instruct": { "id": "microsoft/phi-3-small-128k-instruct", "name": "Phi-3-small instruct (128k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3.5-mini-instruct": { "id": "microsoft/phi-3.5-mini-instruct", "name": "Phi-3.5-mini instruct (128k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-08-20", "last_updated": "2024-08-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3-medium-128k-instruct": { "id": "microsoft/phi-3-medium-128k-instruct", "name": "Phi-3-medium instruct (128k)", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3-small-8k-instruct": { "id": "microsoft/phi-3-small-8k-instruct", "name": "Phi-3-small instruct (8k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-4-reasoning": { "id": "microsoft/phi-4-reasoning", "name": "Phi-4-Reasoning", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/mai-ds-r1": { "id": "microsoft/mai-ds-r1", "name": "MAI-DS-R1", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "mai", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-4-mini-instruct": { "id": "microsoft/phi-4-mini-instruct", "name": "Phi-4-mini-instruct", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-4": { "id": "microsoft/phi-4", "name": "Phi-4", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3.5-vision-instruct": { "id": "microsoft/phi-3.5-vision-instruct", "name": "Phi-3.5-vision instruct (128k)", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-08-20", "last_updated": "2024-08-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3-medium-4k-instruct": { "id": "microsoft/phi-3-medium-4k-instruct", "name": "Phi-3-medium instruct (4k)", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 1024 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-4-multimodal-instruct": { "id": "microsoft/phi-4-multimodal-instruct", "name": "Phi-4-multimodal-instruct", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-4-mini-reasoning": { "id": "microsoft/phi-4-mini-reasoning", "name": "Phi-4-mini-reasoning", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3-mini-128k-instruct": { "id": "microsoft/phi-3-mini-128k-instruct", "name": "Phi-3-mini instruct (128k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-3-mini-4k-instruct": { "id": "microsoft/phi-3-mini-4k-instruct", "name": "Phi-3-mini instruct (4k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 1024 }, "cost": { "input": 0, "output": 0 } }, "openai/o3": { "id": "openai/o3", "name": "OpenAI o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2024-04", "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 0, "output": 0 } }, "openai/o1-mini": { "id": "openai/o1-mini", "name": "OpenAI o1-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2023-10", "release_date": "2024-09-12", "last_updated": "2024-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "OpenAI o4-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2024-04", "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 0, "output": 0 } }, "openai/o1-preview": { "id": "openai/o1-preview", "name": "OpenAI o1-preview", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2023-10", "release_date": "2024-09-12", "last_updated": "2024-09-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "OpenAI o3-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2024-04", "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT-4.1-nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "openai/o1": { "id": "openai/o1", "name": "OpenAI o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2023-10", "release_date": "2024-09-12", "last_updated": "2024-12-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "mistral-ai/mistral-small-2503": { "id": "mistral-ai/mistral-small-2503", "name": "Mistral Small 3.1", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2025-03-01", "last_updated": "2025-03-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "mistral-ai/mistral-nemo": { "id": "mistral-ai/mistral-nemo", "name": "Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mistral-ai/mistral-medium-2505": { "id": "mistral-ai/mistral-medium-2505", "name": "Mistral Medium 3 (25.05)", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2025-05-01", "last_updated": "2025-05-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "mistral-ai/mistral-large-2411": { "id": "mistral-ai/mistral-large-2411", "name": "Mistral Large 24.11", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "mistral-ai/ministral-3b": { "id": "mistral-ai/ministral-3b", "name": "Ministral 3B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mistral-ai/codestral-2501": { "id": "mistral-ai/codestral-2501", "name": "Codestral 25.01", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "codestral", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "cohere/cohere-command-a": { "id": "cohere/cohere-command-a", "name": "Cohere Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "cohere/cohere-command-r-08-2024": { "id": "cohere/cohere-command-r-08-2024", "name": "Cohere Command R 08-2024", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-08-01", "last_updated": "2024-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "cohere/cohere-command-r-plus": { "id": "cohere/cohere-command-r-plus", "name": "Cohere Command R+", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-04-04", "last_updated": "2024-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "cohere/cohere-command-r-plus-08-2024": { "id": "cohere/cohere-command-r-plus-08-2024", "name": "Cohere Command R+ 08-2024", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-08-01", "last_updated": "2024-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "cohere/cohere-command-r": { "id": "cohere/cohere-command-r", "name": "Cohere Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-03-11", "last_updated": "2024-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/meta-llama-3-8b-instruct": { "id": "meta/meta-llama-3-8b-instruct", "name": "Meta-Llama-3-8B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.2-11b-vision-instruct": { "id": "meta/llama-3.2-11b-vision-instruct", "name": "Llama-3.2-11B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/meta-llama-3.1-405b-instruct": { "id": "meta/meta-llama-3.1-405b-instruct", "name": "Meta-Llama-3.1-405B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-4-scout-17b-16e-instruct": { "id": "meta/llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout 17B 16E Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.3-70b-instruct": { "id": "meta/llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "meta/meta-llama-3.1-70b-instruct": { "id": "meta/meta-llama-3.1-70b-instruct", "name": "Meta-Llama-3.1-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "meta/meta-llama-3.1-8b-instruct": { "id": "meta/meta-llama-3.1-8b-instruct", "name": "Meta-Llama-3.1-8B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.2-90b-vision-instruct": { "id": "meta/llama-3.2-90b-vision-instruct", "name": "Llama-3.2-90B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-4-maverick-17b-128e-instruct-fp8": { "id": "meta/llama-4-maverick-17b-128e-instruct-fp8", "name": "Llama 4 Maverick 17B 128E Instruct FP8", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/meta-llama-3-70b-instruct": { "id": "meta/meta-llama-3-70b-instruct", "name": "Meta-Llama-3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "deepseek/deepseek-r1-0528": { "id": "deepseek/deepseek-r1-0528", "name": "DeepSeek-R1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "deepseek/deepseek-v3-0324": { "id": "deepseek/deepseek-v3-0324", "name": "DeepSeek-V3-0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "deepseek/deepseek-r1": { "id": "deepseek/deepseek-r1", "name": "DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 8192 }, "cost": { "input": 0, "output": 0 } } } }, "neuralwatt": { "id": "neuralwatt", "env": ["NEURALWATT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.neuralwatt.com/v1", "name": "Neuralwatt", "doc": "https://portal.neuralwatt.com/docs", "models": { "kimi-k2.5-fast": { "id": "kimi-k2.5-fast", "name": "Kimi K2.5 Fast", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262128, "output": 262128 }, "cost": { "input": 0.52, "output": 2.59, "cache_read": 0.13 } }, "kimi-k2.6-flex": { "id": "kimi-k2.6-flex", "name": "Kimi K2.6 Flex", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262128, "output": 262128 }, "cost": { "input": 0.345, "output": 1.61, "cache_read": 0.08625 } }, "glm-5.2-short-fast-flex": { "id": "glm-5.2-short-fast-flex", "name": "GLM 5.2 Short Fast Flex", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 199984, "output": 199984 }, "cost": { "input": 0.725, "output": 2.25, "cache_read": 0.18125 } }, "glm-5.2-flex": { "id": "glm-5.2-flex", "name": "GLM 5.2 Flex", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048560, "output": 1048560 }, "cost": { "input": 0.725, "output": 2.25, "cache_read": 0.18125 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM 5.2", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048560, "output": 1048560 }, "cost": { "input": 1.45, "output": 4.5, "cache_read": 0.3625 } }, "glm-5.2-short-fast": { "id": "glm-5.2-short-fast", "name": "GLM 5.2 Short Fast", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 199984, "output": 199984 }, "cost": { "input": 1.45, "output": 4.5, "cache_read": 0.3625 } }, "qwen3.5-397b-fast": { "id": "qwen3.5-397b-fast", "name": "Qwen3.5 397B Fast", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262128, "output": 262128 }, "cost": { "input": 0.69, "output": 4.14, "cache_read": 0.1725 } }, "kimi-k2.6-fast": { "id": "kimi-k2.6-fast", "name": "Kimi K2.6 Fast", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262128, "output": 262128 }, "cost": { "input": 0.69, "output": 3.22, "cache_read": 0.1725 } }, "qwen3.6-35b-fast": { "id": "qwen3.6-35b-fast", "name": "Qwen3.6 35B Fast", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "qwen3.6", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131056, "output": 131056 }, "cost": { "input": 0.29, "output": 1.15, "cache_read": 0.0725 } }, "glm-5.2-short-flex": { "id": "glm-5.2-short-flex", "name": "GLM 5.2 Short Flex", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 199984, "output": 199984 }, "cost": { "input": 0.725, "output": 2.25, "cache_read": 0.18125 } }, "glm-5.2-fast": { "id": "glm-5.2-fast", "name": "GLM 5.2 Fast", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048560, "output": 1048560 }, "cost": { "input": 1.45, "output": 4.5, "cache_read": 0.3625 } }, "glm-5.2-short": { "id": "glm-5.2-short", "name": "GLM 5.2 Short", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 199984, "output": 199984 }, "cost": { "input": 1.45, "output": 4.5, "cache_read": 0.3625 } }, "kimi-k2.7-code-flex": { "id": "kimi-k2.7-code-flex", "name": "Kimi K2.7 Code Flex", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.475, "output": 2, "cache_read": 0.11875 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262128, "output": 262128 }, "cost": { "input": 0.69, "output": 3.22, "cache_read": 0.1725 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262128, "output": 262128 }, "cost": { "input": 0.52, "output": 2.59, "cache_read": 0.13 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.2375 } }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131056, "output": 131056 }, "cost": { "input": 0.29, "output": 1.15, "cache_read": 0.0725 } }, "Qwen/Qwen3.5-397B-A17B-FP8": { "id": "Qwen/Qwen3.5-397B-A17B-FP8", "name": "Qwen3.5 397B A17B FP8", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262128, "output": 262128 }, "cost": { "input": 0.69, "output": 4.14, "cache_read": 0.1725 } } } }, "siliconflow-cn": { "id": "siliconflow-cn", "env": ["SILICONFLOW_CN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.siliconflow.cn/v1", "name": "SiliconFlow (China)", "doc": "https://cloud.siliconflow.com/models", "models": { "baidu/ERNIE-4.5-300B-A47B": { "id": "baidu/ERNIE-4.5-300B-A47B", "name": "baidu/ERNIE-4.5-300B-A47B", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "ernie", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-02", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.28, "output": 1.1 } }, "ByteDance-Seed/Seed-OSS-36B-Instruct": { "id": "ByteDance-Seed/Seed-OSS-36B-Instruct", "name": "ByteDance-Seed/Seed-OSS-36B-Instruct", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.21, "output": 0.57 } }, "stepfun-ai/Step-3.5-Flash": { "id": "stepfun-ai/Step-3.5-Flash", "name": "stepfun-ai/Step-3.5-Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "family": "step", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.1, "output": 0.3 } }, "inclusionAI/Ling-flash-2.0": { "id": "inclusionAI/Ling-flash-2.0", "name": "inclusionAI/Ling-flash-2.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-18", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.57 } }, "Pro/moonshotai/Kimi-K2.6": { "id": "Pro/moonshotai/Kimi-K2.6", "name": "Pro/moonshotai/Kimi-K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "Pro/moonshotai/Kimi-K2.5": { "id": "Pro/moonshotai/Kimi-K2.5", "name": "Pro/moonshotai/Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.45, "output": 2.25 } }, "Pro/zai-org/GLM-5": { "id": "Pro/zai-org/GLM-5", "name": "Pro/zai-org/GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 205000, "output": 205000 }, "cost": { "input": 1, "output": 3.2 } }, "Pro/zai-org/GLM-5.1": { "id": "Pro/zai-org/GLM-5.1", "name": "Pro/zai-org/GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-04-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 205000, "output": 205000 }, "cost": { "input": 1.4, "output": 4.4, "cache_write": 0 } }, "Pro/deepseek-ai/DeepSeek-R1": { "id": "Pro/deepseek-ai/DeepSeek-R1", "name": "Pro/deepseek-ai/DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-05-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.5, "output": 2.18 } }, "Pro/deepseek-ai/DeepSeek-V3.1-Terminus": { "id": "Pro/deepseek-ai/DeepSeek-V3.1-Terminus", "name": "Pro/deepseek-ai/DeepSeek-V3.1-Terminus", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 1 } }, "Pro/deepseek-ai/DeepSeek-V3.2": { "id": "Pro/deepseek-ai/DeepSeek-V3.2", "name": "Pro/deepseek-ai/DeepSeek-V3.2", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 0.42 } }, "Pro/deepseek-ai/DeepSeek-V3": { "id": "Pro/deepseek-ai/DeepSeek-V3", "name": "Pro/deepseek-ai/DeepSeek-V3", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-26", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.25, "output": 1 } }, "Pro/MiniMaxAI/MiniMax-M2.5": { "id": "Pro/MiniMaxAI/MiniMax-M2.5", "name": "Pro/MiniMaxAI/MiniMax-M2.5", "description": "Frontier MiniMax model for engineering, office tasks, and agentic reasoning", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 192000, "output": 131000 }, "cost": { "input": 0.3, "output": 1.22 } }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen/Qwen3.6-35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.23, "output": 1.86 } }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen/Qwen3.5-397B-A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.29, "output": 1.74 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen/Qwen3-235B-A22B-Thinking-2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.13, "output": 0.6 } }, "Qwen/Qwen3.5-122B-A10B": { "id": "Qwen/Qwen3.5-122B-A10B", "name": "Qwen/Qwen3.5-122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.29, "output": 2.32 } }, "Qwen/Qwen3.5-27B": { "id": "Qwen/Qwen3.5-27B", "name": "Qwen/Qwen3.5-27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-25", "last_updated": "2026-02-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.26, "output": 2.09 } }, "Qwen/Qwen3-8B": { "id": "Qwen/Qwen3-8B", "name": "Qwen/Qwen3-8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.06, "output": 0.06 } }, "Qwen/Qwen3.5-4B": { "id": "Qwen/Qwen3.5-4B", "name": "Qwen/Qwen3.5-4B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "Qwen/Qwen3.5-9B": { "id": "Qwen/Qwen3.5-9B", "name": "Qwen/Qwen3.5-9B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.22, "output": 1.74 } }, "Qwen/Qwen3-32B": { "id": "Qwen/Qwen3-32B", "name": "Qwen/Qwen3-32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.57 } }, "Qwen/Qwen3-14B": { "id": "Qwen/Qwen3-14B", "name": "Qwen/Qwen3-14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.07, "output": 0.28 } }, "Qwen/Qwen3.5-35B-A3B": { "id": "Qwen/Qwen3.5-35B-A3B", "name": "Qwen/Qwen3.5-35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-25", "last_updated": "2026-02-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.23, "output": 1.86 } }, "Qwen/Qwen3-VL-32B-Thinking": { "id": "Qwen/Qwen3-VL-32B-Thinking", "name": "Qwen/Qwen3-VL-32B-Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-21", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.2, "output": 1.5 } }, "Qwen/Qwen3-VL-30B-A3B-Instruct": { "id": "Qwen/Qwen3-VL-30B-A3B-Instruct", "name": "Qwen/Qwen3-VL-30B-A3B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-05", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.29, "output": 1 } }, "Qwen/Qwen2.5-72B-Instruct": { "id": "Qwen/Qwen2.5-72B-Instruct", "name": "Qwen/Qwen2.5-72B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 33000, "output": 4000 }, "cost": { "input": 0.59, "output": 0.59 } }, "Qwen/Qwen3-VL-32B-Instruct": { "id": "Qwen/Qwen3-VL-32B-Instruct", "name": "Qwen/Qwen3-VL-32B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-21", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.2, "output": 0.6 } }, "Qwen/Qwen3-VL-235B-A22B-Thinking": { "id": "Qwen/Qwen3-VL-235B-A22B-Thinking", "name": "Qwen/Qwen3-VL-235B-A22B-Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-04", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.45, "output": 3.5 } }, "Qwen/Qwen3-VL-8B-Instruct": { "id": "Qwen/Qwen3-VL-8B-Instruct", "name": "Qwen/Qwen3-VL-8B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-15", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.18, "output": 0.68 } }, "Qwen/Qwen3-VL-30B-A3B-Thinking": { "id": "Qwen/Qwen3-VL-30B-A3B-Thinking", "name": "Qwen/Qwen3-VL-30B-A3B-Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-11", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.29, "output": 1 } }, "Qwen/Qwen3-30B-A3B-Instruct-2507": { "id": "Qwen/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen/Qwen3-30B-A3B-Instruct-2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.09, "output": 0.3 } }, "Qwen/Qwen3-Coder-30B-A3B-Instruct": { "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "name": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-01", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.07, "output": 0.28 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-31", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.25, "output": 1 } }, "Qwen/Qwen3-VL-235B-A22B-Instruct": { "id": "Qwen/Qwen3-VL-235B-A22B-Instruct", "name": "Qwen/Qwen3-VL-235B-A22B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-04", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.3, "output": 1.5 } }, "Qwen/Qwen2.5-7B-Instruct": { "id": "Qwen/Qwen2.5-7B-Instruct", "name": "Qwen/Qwen2.5-7B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 33000, "output": 4000 }, "cost": { "input": 0.05, "output": 0.05 } }, "PaddlePaddle/PaddleOCR-VL-1.5": { "id": "PaddlePaddle/PaddleOCR-VL-1.5", "name": "PaddlePaddle/PaddleOCR-VL-1.5", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-29", "last_updated": "2026-01-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "tencent/Hunyuan-A13B-Instruct": { "id": "tencent/Hunyuan-A13B-Instruct", "name": "tencent/Hunyuan-A13B-Instruct", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.57 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1049000, "output": 262000 }, "cost": { "input": 1.4, "output": 4.4, "cache_write": 0 } }, "zai-org/GLM-4.5-Air": { "id": "zai-org/GLM-4.5-Air", "name": "zai-org/GLM-4.5-Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.86 } }, "deepseek-ai/DeepSeek-R1": { "id": "deepseek-ai/DeepSeek-R1", "name": "deepseek-ai/DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-05-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.5, "output": 2.18 } }, "deepseek-ai/DeepSeek-V3.1-Terminus": { "id": "deepseek-ai/DeepSeek-V3.1-Terminus", "name": "deepseek-ai/DeepSeek-V3.1-Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 1 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.003 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "deepseek-ai/DeepSeek-V4-Pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1049000, "output": 393000 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.145 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "deepseek-ai/DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 0.42 } }, "deepseek-ai/DeepSeek-OCR": { "id": "deepseek-ai/DeepSeek-OCR", "name": "deepseek-ai/DeepSeek-OCR", "description": "OCR model for extracting structured text from documents and screenshots", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-20", "last_updated": "2025-10-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "deepseek-ai/DeepSeek-V3": { "id": "deepseek-ai/DeepSeek-V3", "name": "deepseek-ai/DeepSeek-V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-26", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.25, "output": 1 } } } }, "merge-gateway": { "id": "merge-gateway", "env": ["MERGE_GATEWAY_API_KEY"], "npm": "merge-gateway-ai-sdk-provider", "name": "Merge Gateway", "doc": "https://docs.merge.dev/merge-gateway", "models": { "xai/grok-4.3": { "id": "xai/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "xai/grok-4.20-0309-reasoning": { "id": "xai/grok-4.20-0309-reasoning", "name": "Grok 4.20 (Reasoning)", "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "moonshotai/kimi-k2.7-code": { "id": "moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 262144 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 262144 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4 } }, "moonshotai/kimi-k2.7-code-highspeed": { "id": "moonshotai/kimi-k2.7-code-highspeed", "name": "Kimi K2.7 Code Highspeed", "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.9, "output": 8 } }, "mistral/codestral-latest": { "id": "mistral/codestral-latest", "name": "Codestral (latest)", "description": "Mistral code model for completions, refactors, and developer IDE workflows", "family": "codestral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-05-29", "last_updated": "2025-01-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 4096 }, "cost": { "input": 0.3, "output": 0.9 } }, "mistral/mistral-large-latest": { "id": "mistral/mistral-large-latest", "name": "Mistral Large (latest)", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.5, "output": 1.5 } }, "mistral/devstral-small-2507": { "id": "mistral/devstral-small-2507", "name": "Devstral Small", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.3 } }, "mistral/pixtral-large-latest": { "id": "mistral/pixtral-large-latest", "name": "Pixtral Large (latest)", "description": "Mistral's larger vision model for document-heavy image understanding and chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2024-11-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 2, "output": 6 } }, "mistral/mistral-medium-latest": { "id": "mistral/mistral-medium-latest", "name": "Mistral Medium (latest)", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.4, "output": 2 } }, "mistral/mistral-small-latest": { "id": "mistral/mistral-small-latest", "name": "Mistral Small (latest)", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.15, "output": 0.6 } }, "mistral/mistral-medium-2505": { "id": "mistral/mistral-medium-2505", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.4, "output": 2 } }, "mistral/mistral-large-2411": { "id": "mistral/mistral-large-2411", "name": "Mistral Large 2.1", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-18", "last_updated": "2024-11-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 2, "output": 6 } }, "mistral/magistral-medium-latest": { "id": "mistral/magistral-medium-latest", "name": "Magistral Medium (latest)", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-medium", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-03-17", "last_updated": "2025-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2, "output": 5 } }, "mistral/devstral-medium-latest": { "id": "mistral/devstral-medium-latest", "name": "Devstral 2 (latest)", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } }, "mistral/devstral-2512": { "id": "mistral/devstral-2512", "name": "Devstral 2", "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } }, "mistral/mistral-large-2512": { "id": "mistral/mistral-large-2512", "name": "Mistral Large 3", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.5, "output": 1.5 } }, "mistral/devstral-medium-2507": { "id": "mistral/devstral-medium-2507", "name": "Devstral Medium", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 } }, "google/gemini-3.1-pro-preview-customtools": { "id": "google/gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemini-flash-lite-latest": { "id": "google/gemini-flash-lite-latest", "name": "Gemini Flash-Lite Latest", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01, "input_audio": 0.3 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 } }, "google/gemini-3-pro-preview": { "id": "google/gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1 } }, "google/gemini-flash-latest": { "id": "google/gemini-flash-latest", "name": "Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075, "input_audio": 1 } }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "openai/o3": { "id": "openai/o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "openai/gpt-4o-2024-05-13": { "id": "openai/gpt-4o-2024-05-13", "name": "GPT-4o (2024-05-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 5, "output": 15 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "openai/gpt-5-chat-latest": { "id": "openai/gpt-5-chat-latest", "name": "GPT-5 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.3-chat-latest": { "id": "openai/gpt-5.3-chat-latest", "name": "GPT-5.3 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-4o-2024-08-06": { "id": "openai/gpt-4o-2024-08-06", "name": "GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-08-06", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1-chat-latest": { "id": "openai/gpt-5.1-chat-latest", "name": "GPT-5.1 Chat", "description": "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.2-chat-latest": { "id": "openai/gpt-5.2-chat-latest", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "openai/gpt-4o-2024-11-20": { "id": "openai/gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/o1": { "id": "openai/o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 12.5, "output": 75, "cache_read": 1.25 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "zai/glm-4.7": { "id": "zai/glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "zai/glm-4.5": { "id": "zai/glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "zai/glm-4.7-flashx": { "id": "zai/glm-4.7-flashx", "name": "GLM-4.7-FlashX", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.07, "output": 0.4, "cache_read": 0.01, "cache_write": 0 } }, "zai/glm-5.1": { "id": "zai/glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0 } }, "zai/glm-4.6": { "id": "zai/glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "zai/glm-5.2": { "id": "zai/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 50000 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4 } }, "zai/glm-4.5-air": { "id": "zai/glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.2, "output": 1.1, "cache_read": 0.03, "cache_write": 0 } }, "zai/glm-5": { "id": "zai/glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2, "cache_write": 0 } }, "zai/glm-5-turbo": { "id": "zai/glm-5-turbo", "name": "GLM-5-Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24, "cache_write": 0 } }, "anthropic/claude-haiku-4-5-20251001": { "id": "anthropic/claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-opus-4-1-20250805": { "id": "anthropic/claude-opus-4-1-20250805", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4-7": { "id": "anthropic/claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-opus-4-5-20251101": { "id": "anthropic/claude-opus-4-5-20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-opus-4-8": { "id": "anthropic/claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 128000 } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "anthropic/claude-opus-4-20250514": { "id": "anthropic/claude-opus-4-20250514", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-sonnet-4-20250514": { "id": "anthropic/claude-sonnet-4-20250514", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4-6": { "id": "anthropic/claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4-5-20250929": { "id": "anthropic/claude-sonnet-4-5-20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-sonnet-4-6": { "id": "anthropic/claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "cohere/command-a-03-2025": { "id": "cohere/command-a-03-2025", "name": "Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8000 }, "cost": { "input": 2.5, "output": 10 } }, "cohere/command-r-08-2024": { "id": "cohere/command-r-08-2024", "name": "Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.15, "output": 0.6 } }, "cohere/command-r7b-12-2024": { "id": "cohere/command-r7b-12-2024", "name": "Command R7B", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-12-02", "last_updated": "2024-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.0375, "output": 0.15 } }, "cohere/command-r-plus-08-2024": { "id": "cohere/command-r-plus-08-2024", "name": "Command R+", "description": "Cohere's RAG workhorse for long-context enterprise search and tool use", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 2.5, "output": 10 } }, "alibaba/qwen3.7-max": { "id": "alibaba/qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 250000 } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.65, "output": 4.95 } }, "alibaba/qwen3.6-plus": { "id": "alibaba/qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 250000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "minimax/minimax-m2.7-highspeed": { "id": "minimax/minimax-m2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "MiniMax-M2.1", "description": "Earlier MiniMax agent model for practical coding and productivity tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 128000 } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 128000 }, "cost": { "input": 0.6, "output": 2.4 } }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } }, "minimax/minimax-m2.5-highspeed": { "id": "minimax/minimax-m2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } } } }, "qihang-ai": { "id": "qihang-ai", "env": ["QIHANG_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.qhaigc.net/v1", "name": "QiHang", "doc": "https://www.qhaigc.net/docs", "models": { "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-10-01", "last_updated": "2025-10-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.14, "output": 0.71 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.09, "output": 0.71, "tiers": [ { "input": 0.09, "output": 0.71, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 0.09, "output": 0.71 } } }, "claude-opus-4-5-20251101": { "id": "claude-opus-4-5-20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.71, "output": 3.57 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.43, "output": 2.14 } }, "gemini-3-pro-preview": { "id": "gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.57, "output": 3.43 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5-Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0.04, "output": 0.29 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.07, "output": 0.43, "tiers": [ { "input": 0.07, "output": 0.43, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 0.07, "output": 0.43 } } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.14, "output": 1.14 } } } }, "xiaomi-token-plan-ams": { "id": "xiaomi-token-plan-ams", "env": ["XIAOMI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://token-plan-ams.xiaomimimo.com/v1", "name": "Xiaomi Token Plan (Europe)", "doc": "https://platform.xiaomimimo.com/#/docs", "models": { "mimo-v2.5-tts": { "id": "mimo-v2.5-tts", "name": "MiMo-V2.5-TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5-pro": { "id": "mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2-pro": { "id": "mimo-v2-pro", "name": "MiMo-V2-Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 131072 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2-tts": { "id": "mimo-v2-tts", "name": "MiMo-V2-TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5": { "id": "mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2.5-tts-voicedesign": { "id": "mimo-v2.5-tts-voicedesign", "name": "MiMo-V2.5-TTS-VoiceDesign", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5-tts-voiceclone": { "id": "mimo-v2.5-tts-voiceclone", "name": "MiMo-V2.5-TTS-VoiceClone", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } } } }, "modelscope": { "id": "modelscope", "env": ["MODELSCOPE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api-inference.modelscope.cn/v1", "name": "ModelScope", "doc": "https://modelscope.cn/docs/model-service/API-Inference/intro", "models": { "Qwen/Qwen3-30B-A3B-Thinking-2507": { "id": "Qwen/Qwen3-30B-A3B-Thinking-2507", "name": "Qwen3 30B A3B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen3-235B-A22B-Thinking-2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "Qwen/Qwen3-Coder-30B-A3B-Instruct": { "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "name": "Qwen3 Coder 30B A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "Qwen/Qwen3-30B-A3B-Instruct-2507": { "id": "Qwen/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen3 30B A3B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-28", "last_updated": "2025-07-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "ZhipuAI/GLM-4.6": { "id": "ZhipuAI/GLM-4.6", "name": "GLM-4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 98304 }, "cost": { "input": 0, "output": 0 } }, "ZhipuAI/GLM-4.5": { "id": "ZhipuAI/GLM-4.5", "name": "GLM-4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0, "output": 0 } } } }, "groq": { "id": "groq", "env": ["GROQ_API_KEY"], "npm": "@ai-sdk/groq", "name": "Groq", "doc": "https://console.groq.com/docs/models", "models": { "llama-3.3-70b-versatile": { "id": "llama-3.3-70b-versatile", "name": "Llama 3.3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.59, "output": 0.79 } }, "llama-3.1-8b-instant": { "id": "llama-3.1-8b-instant", "name": "Llama 3.1 8B", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.08 } }, "whisper-large-v3-turbo": { "id": "whisper-large-v3-turbo", "name": "Whisper Large V3 Turbo", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 0 } }, "whisper-large-v3": { "id": "whisper-large-v3", "name": "Whisper", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-09-01", "last_updated": "2025-09-05", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 0 } }, "meta-llama/llama-prompt-guard-2-86m": { "id": "meta-llama/llama-prompt-guard-2-86m", "name": "Prompt Guard 2 86M", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-29", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 512 }, "status": "beta", "cost": { "input": 0.04, "output": 0.04 } }, "meta-llama/llama-prompt-guard-2-22m": { "id": "meta-llama/llama-prompt-guard-2-22m", "name": "Llama Prompt Guard 2 22M", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-29", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 512 }, "status": "beta", "cost": { "input": 0.03, "output": 0.03 } }, "meta-llama/llama-4-scout-17b-16e-instruct": { "id": "meta-llama/llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout 17B 16E", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "status": "beta", "cost": { "input": 0.11, "output": 0.34 } }, "openai/gpt-oss-safeguard-20b": { "id": "openai/gpt-oss-safeguard-20b", "name": "Safety GPT OSS 20B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-29", "last_updated": "2026-06-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "status": "beta", "cost": { "input": 0.075, "output": 0.3 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-10-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.075, "output": 0.3, "cache_read": 0.0375 } }, "canopylabs/orpheus-v1-english": { "id": "canopylabs/orpheus-v1-english", "name": "Canopy Labs Orpheus V1 English", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "canopylabs", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-19", "last_updated": "2025-12-19", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 4000, "output": 50000 }, "status": "beta" }, "canopylabs/orpheus-arabic-saudi": { "id": "canopylabs/orpheus-arabic-saudi", "name": "Canopy Labs Orpheus Arabic Saudi", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "canopylabs", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 4000, "output": 50000 }, "status": "beta" }, "groq/compound": { "id": "groq/compound", "name": "Compound", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "groq", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 } }, "groq/compound-mini": { "id": "groq/compound-mini", "name": "Compound Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "groq", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 } }, "qwen/qwen3-32b": { "id": "qwen/qwen3-32b", "name": "Qwen3-32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "default"] } ], "tool_call": true, "temperature": true, "release_date": "2025-06-11", "last_updated": "2025-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 40960 }, "status": "beta", "cost": { "input": 0.29, "output": 0.59 } } } }, "mixlayer": { "id": "mixlayer", "env": ["MIXLAYER_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://models.mixlayer.ai/v1", "name": "Mixlayer", "doc": "https://docs.mixlayer.com", "models": { "qwen/qwen3.5-27b": { "id": "qwen/qwen3.5-27b", "name": "Qwen3.5 27B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.3, "output": 2.4 } }, "qwen/qwen3.5-35b-a3b": { "id": "qwen/qwen3.5-35b-a3b", "name": "Qwen3.5 35B A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.25, "output": 1.3 } }, "qwen/qwen3.5-9b": { "id": "qwen/qwen3.5-9b", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.1, "output": 0.4 } }, "qwen/qwen3.5-397b-a17b": { "id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5 397B A17B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3.6 } }, "qwen/qwen3.5-122b-a10b": { "id": "qwen/qwen3.5-122b-a10b", "name": "Qwen3.5 122B A10B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.4, "output": 3.2 } } } }, "orcarouter": { "id": "orcarouter", "env": ["ORCAROUTER_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.orcarouter.ai/v1", "name": "OrcaRouter", "doc": "https://docs.orcarouter.ai", "models": { "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.13, "output": 0.38 } }, "google/gemini-3.1-pro-preview-customtools": { "id": "google/gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 4, "output": 18, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemini-flash-lite-latest": { "id": "google/gemini-flash-lite-latest", "name": "Gemini Flash-Lite Latest", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01, "input_audio": 0.3 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 4, "output": 18, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.06, "output": 0.33 } }, "google/gemini-3-pro-preview": { "id": "google/gemini-3-pro-preview", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 4, "output": 18, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1 } }, "google/gemini-flash-latest": { "id": "google/gemini-flash-latest", "name": "Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.075, "input_audio": 1 } }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "z-ai/glm-4.7": { "id": "z-ai/glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "z-ai/glm-4.5": { "id": "z-ai/glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "z-ai/glm-5.1": { "id": "z-ai/glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0 } }, "z-ai/glm-4.6": { "id": "z-ai/glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "z-ai/glm-4.5-air": { "id": "z-ai/glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.2, "output": 1.1, "cache_read": 0.03, "cache_write": 0 } }, "z-ai/glm-5": { "id": "z-ai/glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2, "cache_write": 0 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "GPT-5.2 Pro", "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-3.5-turbo": { "id": "openai/gpt-3.5-turbo", "name": "GPT-3.5-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2021-09-01", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 0 } }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/gpt-4": { "id": "openai/gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 30, "output": 60 } }, "openai/gpt-4o-2024-05-13": { "id": "openai/gpt-4o-2024-05-13", "name": "GPT-4o (2024-05-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 5, "output": 15 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "openai/gpt-5-chat-latest": { "id": "openai/gpt-5-chat-latest", "name": "GPT-5 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Codex GPT for repository edits, code review, and practical software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.1-codex-max": { "id": "openai/gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.3-chat-latest": { "id": "openai/gpt-5.3-chat-latest", "name": "GPT-5.3 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-4o-2024-08-06": { "id": "openai/gpt-4o-2024-08-06", "name": "GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-08-06", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-5.1-chat-latest": { "id": "openai/gpt-5.1-chat-latest", "name": "GPT-5.1 Chat", "description": "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.2-chat-latest": { "id": "openai/gpt-5.2-chat-latest", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "openai/gpt-4o-2024-11-20": { "id": "openai/gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 5, "output": 22.5, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "openai/gpt-4-turbo": { "id": "openai/gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 60, "output": 270, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 12.5, "output": 75, "cache_read": 1.25 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "kimi/kimi-k2.5": { "id": "kimi/kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "kimi/kimi-k2.6": { "id": "kimi/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "orcarouter/auto": { "id": "orcarouter/auto", "name": "OrcaRouter Auto", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "auto", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2026-05-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "anthropic/claude-sonnet-4.5": { "id": "anthropic/claude-sonnet-4.5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-opus-4.1": { "id": "anthropic/claude-opus-4.1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.5": { "id": "anthropic/claude-opus-4.5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Claude Opus 4 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "grok/grok-4.3": { "id": "grok/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "qwen/qwen3.6-35b-a3b": { "id": "qwen/qwen3.6-35b-a3b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.248, "output": 1.485 } }, "qwen/qwen3-max": { "id": "qwen/qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.359, "output": 1.434 } }, "qwen/qwen3.5-plus": { "id": "qwen/qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.115, "output": 0.688, "reasoning": 2.4 } }, "qwen/qwen3.5-27b": { "id": "qwen/qwen3.5-27b", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.086, "output": 0.688 } }, "qwen/qwen3.5-35b-a3b": { "id": "qwen/qwen3.5-35b-a3b", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.057, "output": 0.459 } }, "qwen/qwen3.5-397b-a17b": { "id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.172, "output": 1.032 } }, "qwen/qwen3.5-122b-a10b": { "id": "qwen/qwen3.5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.115, "output": 0.917 } }, "qwen/qwen3.6-plus": { "id": "qwen/qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.19, "output": 0.37, "cache_read": 0.0028 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.56, "output": 1.12, "cache_read": 0.003625 } }, "deepseek/deepseek-reasoner": { "id": "deepseek/deepseek-reasoner", "name": "DeepSeek Reasoner", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.028 } }, "deepseek/deepseek-chat": { "id": "deepseek/deepseek-chat", "name": "DeepSeek Chat", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "minimax/minimax-m2.7-highspeed": { "id": "minimax/minimax-m2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } }, "minimax/minimax-m2.5-highspeed": { "id": "minimax/minimax-m2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } } } }, "helicone": { "id": "helicone", "env": ["HELICONE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://ai-gateway.helicone.ai/v1", "name": "Helicone", "doc": "https://helicone.ai/models", "models": { "chatgpt-4o-latest": { "id": "chatgpt-4o-latest", "name": "OpenAI ChatGPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2024-08-14", "last_updated": "2024-08-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 5, "output": 20, "cache_read": 2.5 } }, "gpt-4.1-mini-2025-04-14": { "id": "gpt-4.1-mini-2025-04-14", "name": "OpenAI GPT-4.1 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.39999999999999997, "output": 1.5999999999999999, "cache_read": 0.09999999999999999 } }, "deepseek-v3.1-terminus": { "id": "deepseek-v3.1-terminus", "name": "DeepSeek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.27, "output": 1, "cache_read": 0.21600000000000003 } }, "claude-3.5-haiku": { "id": "claude-3.5-haiku", "name": "Anthropic: Claude 3.5 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 0.7999999999999999, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "llama-3.1-8b-instruct": { "id": "llama-3.1-8b-instruct", "name": "Meta Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.02, "output": 0.049999999999999996 } }, "o3": { "id": "o3", "name": "OpenAI o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2024-06", "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "llama-prompt-guard-2-86m": { "id": "llama-prompt-guard-2-86m", "name": "Meta Llama Prompt Guard 2 86M", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 512, "output": 2 }, "cost": { "input": 0.01, "output": 0.01 } }, "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3 Coder 30B A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.09999999999999999, "output": 0.3 } }, "hermes-2-pro-llama-3-8b": { "id": "hermes-2-pro-llama-3-8b", "name": "Hermes 2 Pro Llama 3 8B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-05", "release_date": "2024-05-27", "last_updated": "2024-05-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.14, "output": 0.14 } }, "deepseek-v3": { "id": "deepseek-v3", "name": "DeepSeek V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-26", "last_updated": "2024-12-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.56, "output": 1.68, "cache_read": 0.07 } }, "grok-code-fast-1": { "id": "grok-code-fast-1", "name": "xAI Grok Code Fast 1", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2024-08-25", "last_updated": "2024-08-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 10000 }, "cost": { "input": 0.19999999999999998, "output": 1.5, "cache_read": 0.02 } }, "o1-mini": { "id": "o1-mini", "name": "OpenAI: o1-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 65536 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "deepseek-r1-distill-llama-70b": { "id": "deepseek-r1-distill-llama-70b", "name": "DeepSeek R1 Distill Llama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.03, "output": 0.13 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 40960 }, "cost": { "input": 0.29, "output": 0.59 } }, "claude-sonnet-4": { "id": "claude-sonnet-4", "name": "Anthropic: Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-14", "last_updated": "2025-05-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.30000000000000004, "cache_write": 3.75 } }, "qwen3-next-80b-a3b-instruct": { "id": "qwen3-next-80b-a3b-instruct", "name": "Qwen3 Next 80B A3B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 16384 }, "cost": { "input": 0.14, "output": 1.4 } }, "llama-4-maverick": { "id": "llama-4-maverick", "name": "Meta Llama 4 Maverick 17B 128E", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.15, "output": 0.6 } }, "mistral-nemo": { "id": "mistral-nemo", "name": "Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16400 }, "cost": { "input": 20, "output": 40 } }, "llama-3.3-70b-versatile": { "id": "llama-3.3-70b-versatile", "name": "Meta Llama 3.3 70B Versatile", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32678 }, "cost": { "input": 0.59, "output": 0.7899999999999999 } }, "gemma2-9b-it": { "id": "gemma2-9b-it", "name": "Google Gemma 2", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-06", "release_date": "2024-06-25", "last_updated": "2024-06-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.01, "output": 0.03 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Google Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.3125, "cache_write": 1.25 } }, "gpt-5": { "id": "gpt-5", "name": "OpenAI GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12500000000000003 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Anthropic: Claude 4.5 Haiku (20251001)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-10", "release_date": "2025-10-01", "last_updated": "2025-10-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 1, "output": 5, "cache_read": 0.09999999999999999, "cache_write": 1.25 } }, "claude-4.5-haiku": { "id": "claude-4.5-haiku", "name": "Anthropic: Claude 4.5 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-10", "release_date": "2025-10-01", "last_updated": "2025-10-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 1, "output": 5, "cache_read": 0.09999999999999999, "cache_write": 1.25 } }, "gpt-5-pro": { "id": "gpt-5-pro", "name": "OpenAI: GPT-5 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 15, "output": 120 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Google Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075, "cache_write": 0.3 } }, "gpt-4o": { "id": "gpt-4o", "name": "OpenAI GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-05", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "o4-mini": { "id": "o4-mini", "name": "OpenAI o4 Mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2024-06", "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "llama-3.1-8b-instant": { "id": "llama-3.1-8b-instant", "name": "Meta Llama 3.1 8B Instant", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32678 }, "cost": { "input": 0.049999999999999996, "output": 0.08 } }, "llama-prompt-guard-2-22m": { "id": "llama-prompt-guard-2-22m", "name": "Meta Llama Prompt Guard 2 22M", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 512, "output": 2 }, "cost": { "input": 0.01, "output": 0.01 } }, "o3-pro": { "id": "o3-pro", "name": "OpenAI o3 Pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2024-06", "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 20, "output": 80 } }, "grok-3-mini": { "id": "grok-3-mini", "name": "xAI Grok 3 Mini", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.3, "output": 0.5, "cache_read": 0.075 } }, "claude-opus-4-1-20250805": { "id": "claude-opus-4-1-20250805", "name": "Anthropic: Claude Opus 4.1 (20250805)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "deepseek-tng-r1t2-chimera": { "id": "deepseek-tng-r1t2-chimera", "name": "DeepSeek TNG R1T2 Chimera", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek-thinking", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-02", "last_updated": "2025-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 130000, "output": 163840 }, "cost": { "input": 0.3, "output": 1.2 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Meta Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16400 }, "cost": { "input": 0.13, "output": 0.39 } }, "sonar-reasoning-pro": { "id": "sonar-reasoning-pro", "name": "Perplexity Sonar Reasoning Pro", "description": "Web-grounded reasoning model for multi-step research and cited answers", "family": "sonar-reasoning", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "output": 4096 }, "cost": { "input": 2, "output": 8 } }, "grok-3": { "id": "grok-3", "name": "xAI Grok 3", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75 } }, "glm-4.6": { "id": "glm-4.6", "name": "Zai GLM-4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.44999999999999996, "output": 1.5 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 262144 }, "cost": { "input": 0.48, "output": 2 } }, "grok-4-1-fast-non-reasoning": { "id": "grok-4-1-fast-non-reasoning", "name": "xAI Grok 4.1 Fast Non-Reasoning", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-17", "last_updated": "2025-11-17", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.19999999999999998, "output": 0.5, "cache_read": 0.049999999999999996 } }, "qwen3-coder": { "id": "qwen3-coder", "name": "Qwen3 Coder 480B A35B Instruct Turbo", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.22, "output": 0.95 } }, "gpt-5-chat-latest": { "id": "gpt-5-chat-latest", "name": "OpenAI GPT-5 Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2024-09", "release_date": "2024-09-30", "last_updated": "2024-09-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12500000000000003 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "OpenAI: GPT-5.1 Codex", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-codex", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12500000000000003 } }, "o3-mini": { "id": "o3-mini", "name": "OpenAI o3 Mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2023-10", "release_date": "2023-10-01", "last_updated": "2023-10-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "mistral-large-2411": { "id": "mistral-large-2411", "name": "Mistral-Large", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-24", "last_updated": "2024-07-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 2, "output": 6 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "Google Gemini 2.5 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.09999999999999999, "output": 0.39999999999999997, "cache_read": 0.024999999999999998, "cache_write": 0.09999999999999999 } }, "llama-guard-4": { "id": "llama-guard-4", "name": "Meta Llama Guard 4 12B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 1024 }, "cost": { "input": 0.21, "output": 0.21 } }, "claude-opus-4-1": { "id": "claude-opus-4-1", "name": "Anthropic: Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "claude-3.7-sonnet": { "id": "claude-3.7-sonnet", "name": "Anthropic: Claude 3.7 Sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-02", "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.30000000000000004, "cache_write": 3.75 } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "OpenAI: GPT-5.1 Codex Mini", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-codex", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.024999999999999998 } }, "gpt-5.1-chat-latest": { "id": "gpt-5.1-chat-latest", "name": "OpenAI GPT-5.1 Chat", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-codex", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12500000000000003 } }, "claude-3-haiku-20240307": { "id": "claude-3-haiku-20240307", "name": "Anthropic: Claude 3 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-03-07", "last_updated": "2024-03-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.03, "cache_write": 0.3 } }, "grok-4-fast-reasoning": { "id": "grok-4-fast-reasoning", "name": "xAI: Grok 4 Fast Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.19999999999999998, "output": 0.5, "cache_read": 0.049999999999999996 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "OpenAI GPT-4.1 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.09999999999999999, "output": 0.39999999999999997, "cache_read": 0.024999999999999998 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "OpenAI GPT-OSS 120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.04, "output": 0.16 } }, "sonar": { "id": "sonar", "name": "Perplexity Sonar", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "output": 4096 }, "cost": { "input": 1, "output": 1 } }, "qwen2.5-coder-7b-fast": { "id": "qwen2.5-coder-7b-fast", "name": "Qwen2.5 Coder 7B fast", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-09", "release_date": "2024-09-15", "last_updated": "2024-09-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 8192 }, "cost": { "input": 0.03, "output": 0.09 } }, "o1": { "id": "o1", "name": "OpenAI: o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "ernie-4.5-21b-a3b-thinking": { "id": "ernie-4.5-21b-a3b-thinking", "name": "Baidu Ernie 4.5 21B A3B Thinking", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "ernie", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-03", "release_date": "2025-03-16", "last_updated": "2025-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8000 }, "cost": { "input": 0.07, "output": 0.28 } }, "llama-4-scout": { "id": "llama-4-scout", "name": "Meta Llama 4 Scout 17B 16E", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.08, "output": 0.3 } }, "sonar-pro": { "id": "sonar-pro", "name": "Perplexity Sonar Pro", "description": "Advanced Sonar search model for deeper research and cited synthesis", "family": "sonar-pro", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 3, "output": 15 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "OpenAI GPT-4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "Anthropic: Claude Sonnet 4.5 (20250929)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.30000000000000004, "cache_write": 3.75 } }, "deepseek-reasoner": { "id": "deepseek-reasoner", "name": "DeepSeek Reasoner", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0.56, "output": 1.68, "cache_read": 0.07 } }, "grok-4-1-fast-reasoning": { "id": "grok-4-1-fast-reasoning", "name": "xAI Grok 4.1 Fast Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-17", "last_updated": "2025-11-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.19999999999999998, "output": 0.5, "cache_read": 0.049999999999999996 } }, "gemini-3-pro-preview": { "id": "gemini-3-pro-preview", "name": "Google Gemini 3 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.19999999999999998 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "OpenAI GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.024999999999999998 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "OpenAI GPT-4.1 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.39999999999999997, "output": 1.5999999999999999, "cache_read": 0.09999999999999999 } }, "sonar-reasoning": { "id": "sonar-reasoning", "name": "Perplexity Sonar Reasoning", "description": "Web-grounded reasoning model for multi-step research and cited answers", "family": "sonar-reasoning", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "output": 4096 }, "cost": { "input": 1, "output": 5 } }, "sonar-deep-research": { "id": "sonar-deep-research", "name": "Perplexity Sonar Deep Research", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar-deep-research", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "output": 4096 }, "cost": { "input": 2, "output": 8 } }, "kimi-k2-0905": { "id": "kimi-k2-0905", "name": "Kimi K2 (09/05)", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.5, "output": 2, "cache_read": 0.39999999999999997 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "OpenAI GPT-5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.049999999999999996, "output": 0.39999999999999997, "cache_read": 0.005 } }, "grok-4": { "id": "grok-4", "name": "xAI Grok 4", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-09", "last_updated": "2024-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75 } }, "qwen3-235b-a22b-thinking": { "id": "qwen3-235b-a22b-thinking", "name": "Qwen3 235B A22B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 81920 }, "cost": { "input": 0.3, "output": 2.9000000000000004 } }, "claude-opus-4": { "id": "claude-opus-4", "name": "Anthropic: Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-14", "last_updated": "2025-05-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "grok-4-fast-non-reasoning": { "id": "grok-4-fast-non-reasoning", "name": "xAI Grok 4 Fast Non-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.19999999999999998, "output": 0.5, "cache_read": 0.049999999999999996 } }, "claude-4.5-opus": { "id": "claude-4.5-opus", "name": "Anthropic: Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "qwen3-vl-235b-a22b-instruct": { "id": "qwen3-vl-235b-a22b-instruct", "name": "Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.3, "output": 1.5 } }, "kimi-k2-0711": { "id": "kimi-k2-0711", "name": "Kimi K2 (07/11)", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.5700000000000001, "output": 2.3 } }, "gemma-3-12b-it": { "id": "gemma-3-12b-it", "name": "Google Gemma 3 12B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.049999999999999996, "output": 0.09999999999999999 } }, "gpt-4o-mini": { "id": "gpt-4o-mini", "name": "OpenAI GPT-4o-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "OpenAI GPT-OSS 20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.049999999999999996, "output": 0.19999999999999998 } }, "claude-3.5-sonnet-v2": { "id": "claude-3.5-sonnet-v2", "name": "Anthropic: Claude 3.5 Sonnet v2", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 3, "output": 15, "cache_read": 0.30000000000000004, "cache_write": 3.75 } }, "qwen3-30b-a3b": { "id": "qwen3-30b-a3b", "name": "Qwen3 30B A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 41000, "output": 41000 }, "cost": { "input": 0.08, "output": 0.29 } }, "gpt-5-codex": { "id": "gpt-5-codex", "name": "OpenAI: GPT-5 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12500000000000003 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.27, "output": 0.41 } }, "claude-4.5-sonnet": { "id": "claude-4.5-sonnet", "name": "Anthropic: Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.30000000000000004, "cache_write": 3.75 } }, "mistral-small": { "id": "mistral-small", "name": "Mistral Small 3.2", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.075, "output": 0.2 } }, "llama-3.1-8b-instruct-turbo": { "id": "llama-3.1-8b-instruct-turbo", "name": "Meta Llama 3.1 8B Instruct Turbo", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.02, "output": 0.03 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "OpenAI GPT-5.1", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": false, "knowledge": "2025-01", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.12500000000000003 } } } }, "zai": { "id": "zai", "env": ["ZHIPU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.z.ai/api/paas/v4", "name": "Z.AI", "doc": "https://docs.z.ai/guides/overview/pricing", "models": { "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "glm-4.5v": { "id": "glm-4.5v", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8 } }, "glm-4.5": { "id": "glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "glm-4.7-flashx": { "id": "glm-4.7-flashx", "name": "GLM-4.7-FlashX", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.07, "output": 0.4, "cache_read": 0.01, "cache_write": 0 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0 } }, "glm-4.6": { "id": "glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0 } }, "glm-4.6v": { "id": "glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9 } }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24, "cache_write": 0 } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.2, "output": 1.1, "cache_read": 0.03, "cache_write": 0 } }, "glm-4.7-flash": { "id": "glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.5-flash": { "id": "glm-4.5-flash", "name": "GLM-4.5-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2, "cache_write": 0 } }, "glm-5-turbo": { "id": "glm-5-turbo", "name": "GLM-5-Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24, "cache_write": 0 } } } }, "nearai": { "id": "nearai", "env": ["NEARAI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://cloud-api.near.ai/v1", "name": "NEAR AI Cloud", "doc": "https://docs.near.ai/", "models": { "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.13, "output": 0.4, "cache_read": 0.026 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "google/gemini-3-pro": { "id": "google/gemini-3-pro", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 15, "cache_read": 0 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01, "input_audio": 0.3 } }, "Qwen/Qwen3-Embedding-0.6B": { "id": "Qwen/Qwen3-Embedding-0.6B", "name": "Qwen3 Embedding 0.6B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-06-03", "last_updated": "2025-06-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 1024 }, "cost": { "input": 0.01, "output": 0 } }, "Qwen/Qwen3-Reranker-0.6B": { "id": "Qwen/Qwen3-Reranker-0.6B", "name": "Qwen3 Reranker 0.6B", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-06-03", "last_updated": "2025-06-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 1024 }, "cost": { "input": 0.01, "output": 0.01 } }, "Qwen/Qwen3.6-35B-A3B-FP8": { "id": "Qwen/Qwen3.6-35B-A3B-FP8", "name": "Qwen 3.6 35B A3B FP8", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.17, "output": 1.1, "cache_read": 0.056 } }, "Qwen/Qwen3.5-122B-A10B": { "id": "Qwen/Qwen3.5-122B-A10B", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.4, "output": 3.2 } }, "Qwen/Qwen3-30B-A3B-Instruct-2507": { "id": "Qwen/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen3 30B-A3B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.15, "output": 0.55 } }, "Qwen/Qwen3-VL-30B-A3B-Instruct": { "id": "Qwen/Qwen3-VL-30B-A3B-Instruct", "name": "Qwen3-VL 30B-A3B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.55 } }, "openai/o3": { "id": "openai/o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.8, "output": 15.5, "cache_read": 0.18 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT-OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.55 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "openai/whisper-large-v3": { "id": "openai/whisper-large-v3", "name": "Whisper Large v3", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-11-06", "last_updated": "2023-11-06", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 448, "output": 448 }, "cost": { "input": 0.01, "output": 0 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "anthropic/claude-sonnet-4-5": { "id": "anthropic/claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15.5, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4-7": { "id": "anthropic/claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-haiku-4-5": { "id": "anthropic/claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-opus-4-6": { "id": "anthropic/claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4-6": { "id": "anthropic/claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "zai-org/GLM-5.1-FP8": { "id": "zai-org/GLM-5.1-FP8", "name": "GLM-5.1 FP8", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.85, "output": 3.3 } }, "black-forest-labs/FLUX.2-klein-4B": { "id": "black-forest-labs/FLUX.2-klein-4B", "name": "FLUX.2 Klein 4B", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 1, "output": 1 } } } }, "pioneer": { "id": "pioneer", "env": ["PIONEER_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.pioneer.ai/v1", "name": "Pioneer", "doc": "https://agent.pioneer.ai/llms.txt", "models": { "gemini-3-flash": { "id": "gemini-3-flash", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.083333 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.32, "output": 1.28, "cache_read": 0.064, "cache_write": 0.4 } }, "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.25, "output": 3.75, "cache_read": 0.25, "cache_write": 1.5625 } }, "gpt-4o": { "id": "gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25, "cache_write": 2.5 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "cache_write": 0.083333 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02, "cache_write": 0.2 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1875, "output": 1.125, "cache_read": 0.0375, "cache_write": 0.234375 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "mistral-medium-3.5": { "id": "mistral-medium-3.5", "name": "Mistral Medium 3.5", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 1.5, "output": 7.5, "cache_read": 1.5, "cache_write": 1.5 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175, "cache_write": 1.75 } }, "claude-opus-4-1": { "id": "claude-opus-4-1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.05, "cache_write": 0.1 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 2.5 } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075, "cache_write": 0.75 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 1, "cache_write": 2 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025, "cache_write": 0.25 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.2, "cache_write": 0.4 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005, "cache_write": 0.05 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemini-3.1-pro": { "id": "gemini-3.1-pro", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "cache_write": 0.375 } }, "gpt-4o-mini": { "id": "gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075, "cache_write": 0.15 } }, "qwen3.6-max-preview": { "id": "qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.04, "output": 6.24, "cache_read": 0.208, "cache_write": 1.3 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.325, "output": 1.95, "cache_read": 0.065, "cache_write": 0.40625 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "cache_write": 1.25 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 5 } }, "LiquidAI/LFM2-24B-A2B": { "id": "LiquidAI/LFM2-24B-A2B", "name": "LFM2 24B A2B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "liquid", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-01-31", "last_updated": "2026-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.03, "output": 0.12, "cache_read": 0.03, "cache_write": 0.03 } }, "HuggingFaceTB/SmolLM3-3B-Base": { "id": "HuggingFaceTB/SmolLM3-3B-Base", "name": "SmolLM3 3B Base", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "meta-llama/Llama-3.2-1B-Instruct": { "id": "meta-llama/Llama-3.2-1B-Instruct", "name": "Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-31", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 60000 }, "cost": { "input": 0.1, "output": 0.201, "cache_read": 0.1, "cache_write": 0.1 } }, "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 4096 }, "cost": { "input": 0.9, "output": 0.9, "cache_read": 0.9, "cache_write": 0.9 } }, "meta-llama/Llama-3.1-8B-Instruct": { "id": "meta-llama/Llama-3.1-8B-Instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-06-30", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.2, "output": 0.2, "cache_read": 0.2, "cache_write": 0.2 } }, "meta-llama/Llama-3.2-3B-Instruct": { "id": "meta-llama/Llama-3.2-3B-Instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-31", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 80000 }, "cost": { "input": 0.1, "output": 0.335, "cache_read": 0.1, "cache_write": 0.1 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.34, "cache_write": 0.95 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19, "cache_write": 0.95 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.5, "output": 0.5, "cache_read": 0.5, "cache_write": 0.5 } }, "google/diffusiongemma-26B-A4B-it": { "id": "google/diffusiongemma-26B-A4B-it", "name": "DiffusionGemma 26B-A4B IT", "description": "Gemini model for general assistance, reasoning, and multimodal workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-05-31", "last_updated": "2026-05-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 4096 }, "cost": { "input": 0.5, "output": 0.5, "cache_read": 0.5, "cache_write": 0.5 } }, "google/gemma-3-4b-pt": { "id": "google/gemma-3-4b-pt", "name": "Gemma 3 4B (Pretrained)", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-02-28", "last_updated": "2025-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "google/gemma-4-E4B-it": { "id": "google/gemma-4-E4B-it", "name": "Gemma 4 E4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.2, "output": 0.2, "cache_read": 0.2, "cache_write": 0.2 } }, "google/gemma-4-E2B-it": { "id": "google/gemma-4-E2B-it", "name": "Gemma 4 E2B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.1, "output": 0.1, "cache_read": 0.1, "cache_write": 0.1 } }, "google/gemma-4-12B-it": { "id": "google/gemma-4-12B-it", "name": "Gemma 4 12B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-05-31", "last_updated": "2026-05-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.25, "output": 0.25, "cache_read": 0.25, "cache_write": 0.25 } }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.14, "output": 1, "cache_read": 0.028, "cache_write": 0.175 } }, "Qwen/Qwen3-4B-Instruct-2507": { "id": "Qwen/Qwen3-4B-Instruct-2507", "name": "Qwen3 4B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 4096 }, "cost": { "input": 0.2, "output": 0.2, "cache_read": 0.2, "cache_write": 0.2 } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.6, "output": 0.6, "cache_read": 0.6, "cache_write": 0.6 } }, "Qwen/Qwen3-4B-Base": { "id": "Qwen/Qwen3-4B-Base", "name": "Qwen3 4B Base", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-03-31", "last_updated": "2025-03-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "Qwen/Qwen3-1.7B-Base": { "id": "Qwen/Qwen3-1.7B-Base", "name": "Qwen3 1.7B Base", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-03-31", "last_updated": "2025-03-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.1, "output": 0.1, "cache_read": 0.1, "cache_write": 0.1 } }, "Qwen/Qwen3-8B": { "id": "Qwen/Qwen3-8B", "name": "Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-03-31", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 4096 }, "cost": { "input": 0.2, "output": 0.2, "cache_read": 0.2, "cache_write": 0.2 } }, "Qwen/Qwen3.5-9B": { "id": "Qwen/Qwen3.5-9B", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.3, "output": 0.3, "cache_read": 0.3, "cache_write": 0.3 } }, "Qwen/Qwen3-32B": { "id": "Qwen/Qwen3-32B", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.9, "output": 0.9, "cache_read": 0.9, "cache_write": 0.9 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015, "cache_write": 0.15 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.07, "output": 0.3, "cache_read": 0.035, "cache_write": 0.07 } }, "XiaomiMiMo/MiMo-V2.5": { "id": "XiaomiMiMo/MiMo-V2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1050000, "output": 131072 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028, "cache_write": 0.14 } }, "XiaomiMiMo/MiMo-V2.5-Pro": { "id": "XiaomiMiMo/MiMo-V2.5-Pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1050000, "output": 131072 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036, "cache_write": 0.435 } }, "mistralai/Mistral-Small-4-119B-2603": { "id": "mistralai/Mistral-Small-4-119B-2603", "name": "Mistral Small 4", "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015, "cache_write": 0.15 } }, "mistralai/Mistral-7B-Instruct-v0.3": { "id": "mistralai/Mistral-7B-Instruct-v0.3", "name": "Mistral 7B Instruct v0.3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2023-04-30", "last_updated": "2023-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.2, "output": 0.2, "cache_read": 0.2, "cache_write": 0.2 } }, "mistralai/Mistral-Nemo-Instruct-2407": { "id": "mistralai/Mistral-Nemo-Instruct-2407", "name": "Mistral Nemo", "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 128000 }, "cost": { "input": 0.02, "output": 0.03, "cache_read": 0.02, "cache_write": 0.02 } }, "pioneer/auto": { "id": "pioneer/auto", "name": "Pioneer Auto", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "auto", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2024-01-01", "last_updated": "2025-06-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 4096 } }, "sakana/fugu-ultra": { "id": "sakana/fugu-ultra", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "family": "fugu", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "release_date": "2026-06-15", "last_updated": "2026-06-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 5 } }, "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16": { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65000 }, "cost": { "input": 0.5, "output": 2.5, "cache_read": 0.15, "cache_write": 0.5 } }, "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16": { "id": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16", "name": "Nemotron 3 Nano 30B A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.05, "output": 0.2, "cache_read": 0.05, "cache_write": 0.05 } }, "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8": { "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", "name": "Nemotron 3 Super 120B A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 0.09, "output": 0.45, "cache_read": 0.09, "cache_write": 0.09 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 128000 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 1.4 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.98, "output": 3.08, "cache_read": 0.182, "cache_write": 0.98 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.1, "output": 0.2, "cache_read": 0.0197, "cache_write": 0.1 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625, "cache_write": 0.435 } }, "fastino/gliner2-multi-large-v1": { "id": "fastino/gliner2-multi-large-v1", "name": "GLiNER2 Multi Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-11-30", "last_updated": "2025-11-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "fastino/gliner2-multi-v1": { "id": "fastino/gliner2-multi-v1", "name": "GLiNER2 Multi", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-11-30", "last_updated": "2025-11-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "fastino/gliguard-LLMGuardrails-300M": { "id": "fastino/gliguard-LLMGuardrails-300M", "name": "GLiGuard LLM Guardrails 300M", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "fastino/gliner2-privacy-filter-PII-multi": { "id": "fastino/gliner2-privacy-filter-PII-multi", "name": "GLiNER2 Privacy Filter PII (Multi)", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "fastino/gliner2-base-v1": { "id": "fastino/gliner2-base-v1", "name": "GLiNER2 Base", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "fastino/gliner2-large-v1": { "id": "fastino/gliner2-large-v1", "name": "GLiNER2 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.15, "cache_write": 0.15 } }, "MiniMaxAI/MiniMax-M3": { "id": "MiniMaxAI/MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.3 } }, "MiniMaxAI/MiniMax-M2.7": { "id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.279, "output": 1.2, "cache_read": 0.279, "cache_write": 0.279 } } } }, "llmgateway": { "id": "llmgateway", "env": ["LLMGATEWAY_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.llmgateway.io/v1", "name": "LLM Gateway", "doc": "https://llmgateway.io/docs", "models": { "qwen-coder-plus": { "id": "qwen-coder-plus", "name": "Qwen Coder Plus", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.502, "output": 1.004 } }, "mistral-large-latest": { "id": "mistral-large-latest", "name": "Mistral Large (latest)", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 262144 }, "cost": { "input": 4, "output": 12 } }, "qwen3-vl-235b-a22b-thinking": { "id": "qwen3-vl-235b-a22b-thinking", "name": "Qwen3 VL 235B A22B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.98, "output": 3.95 } }, "devstral-small-2507": { "id": "devstral-small-2507", "name": "Devstral Small", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.3 } }, "qwen3-vl-30b-a3b-thinking": { "id": "qwen3-vl-30b-a3b-thinking", "name": "Qwen3 VL 30B A3B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-02", "last_updated": "2025-10-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.2, "output": 1 } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1050000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "qwen3-coder-plus": { "id": "qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Hosted Qwen coder for software agents, repo edits, and long-context code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 6, "output": 60, "cache_read": 1.2, "cache_write": 7.5 } }, "minimax-m2.7-highspeed": { "id": "minimax-m2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "qwen-plus": { "id": "qwen-plus", "name": "Qwen Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.4, "output": 1.2, "reasoning": 4, "cache_read": 0.08, "cache_write": 0.5 } }, "o3": { "id": "o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "nemotron-3-ultra-550b": { "id": "nemotron-3-ultra-550b", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 128000 }, "cost": { "input": 0.5, "output": 2.5, "cache_read": 0.15 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 228700, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "grok-4-20-beta-0309-non-reasoning": { "id": "grok-4-20-beta-0309-non-reasoning", "name": "Grok 4.20 (Non-Reasoning)", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.38, "output": 1.98, "cache_read": 0.19, "cache_write": 0 } }, "gemini-3.1-flash-lite": { "id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "cache_write": 0.08333 } }, "qwen3-235b-a22b-instruct-2507": { "id": "qwen3-235b-a22b-instruct-2507", "name": "Qwen3 235B A22B Instruct (2507)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-08", "last_updated": "2025-07-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.09, "output": 0.58 } }, "llama-3-70b-instruct": { "id": "llama-3-70b-instruct", "name": "Llama 3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8000 }, "cost": { "input": 0.51, "output": 0.74 } }, "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 65536 }, "cost": { "input": 0.07, "output": 0.27 } }, "seed-1-8-251228": { "id": "seed-1-8-251228", "name": "Seed 1.8 (251228)", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-18", "last_updated": "2025-12-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.05 } }, "kimi-k2": { "id": "kimi-k2", "name": "Kimi K2", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-11", "last_updated": "2025-07-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.57, "output": 2.3, "cache_read": 0.5 } }, "llama-3.1-70b-instruct": { "id": "llama-3.1-70b-instruct", "name": "Llama 3.1 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 2048 }, "status": "beta", "cost": { "input": 0.72, "output": 0.72 } }, "gpt-5.2-pro": { "id": "gpt-5.2-pro", "name": "GPT-5.2 Pro", "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "minimax-m2.1": { "id": "minimax-m2.1", "name": "MiniMax-M2.1", "description": "Earlier MiniMax agent model for practical coding and productivity tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.27, "output": 1.1 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.1, "output": 0.3, "reasoning": 8.4 } }, "pixtral-large-latest": { "id": "pixtral-large-latest", "name": "Pixtral Large (latest)", "description": "Mistral's larger vision model for document-heavy image understanding and chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2024-11-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 4, "output": 12 } }, "glm-4-32b-0414-128k": { "id": "glm-4-32b-0414-128k", "name": "GLM-4 32B (0414-128k)", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.1, "output": 0.1 } }, "seed-1-6-flash-250715": { "id": "seed-1-6-flash-250715", "name": "Seed 1.6 Flash (250715)", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.07, "output": 0.3, "cache_read": 0.015 } }, "qwen3-next-80b-a3b-instruct": { "id": "qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 1.2 } }, "qwen3-vl-8b-instruct": { "id": "qwen3-vl-8b-instruct", "name": "Qwen3 VL 8B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.08, "output": 0.5 } }, "gpt-4o-mini-search-preview": { "id": "gpt-4o-mini-search-preview", "name": "GPT-4o Mini Search Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "glm-4.5v": { "id": "glm-4.5v", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8, "cache_read": 0.11 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.08, "cache_write": 0.5 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "qwen3.6-35b-a3b": { "id": "qwen3.6-35b-a3b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.248, "output": 1.485 } }, "gpt-3.5-turbo": { "id": "gpt-3.5-turbo", "name": "GPT-3.5-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2021-09-01", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 0 } }, "seed-1-6-250615": { "id": "seed-1-6-250615", "name": "Seed 1.6 (250615)", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-06-25", "last_updated": "2025-06-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.05 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.25, "output": 3.75, "cache_read": 0.125, "cache_write": 3.125 } }, "glm-4.5": { "id": "glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "gpt-5-pro": { "id": "gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] }, { "type": "budget_tokens", "min": 1, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03 } }, "gpt-4o": { "id": "gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "ministral-8b-2512": { "id": "ministral-8b-2512", "name": "Ministral 8B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.15, "output": 0.15 } }, "gpt-4": { "id": "gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 30, "output": 60 } }, "o4-mini": { "id": "o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "qwen3-max": { "id": "qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.845, "output": 3.38, "cache_read": 0.6, "cache_write": 3.75 } }, "mistral-small-2506": { "id": "mistral-small-2506", "name": "Mistral Small 3.2", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.1, "output": 0.3 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "cache_write": 0.08333 } }, "qwen-plus-latest": { "id": "qwen-plus-latest", "name": "Qwen Plus Latest", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-01-25", "last_updated": "2025-01-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 8192 }, "cost": { "input": 0.4, "output": 1.2, "cache_read": 0.08, "cache_write": 0.5 } }, "muse-spark-1.1": { "id": "muse-spark-1.1", "name": "Muse Spark 1.1", "description": "Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.", "family": "muse", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 32000 }, "cost": { "input": 1.25, "output": 4.25, "cache_read": 0.15 } }, "claude-opus-4-1-20250805": { "id": "claude-opus-4-1-20250805", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "gemma-4-31b-it": { "id": "gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.13, "output": 0.38 } }, "grok-4-3": { "id": "grok-4-3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.3125, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "llama-4-scout-17b-instruct": { "id": "llama-4-scout-17b-instruct", "name": "Llama 4 Scout 17B Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 2048 }, "cost": { "input": 0.18, "output": 0.59 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 4096 }, "cost": { "input": 0.13, "output": 0.4 } }, "sonar-reasoning-pro": { "id": "sonar-reasoning-pro", "name": "Sonar Reasoning Pro", "description": "Web-grounded Sonar for multi-step research questions that need cited reasoning", "family": "sonar-reasoning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 2, "output": 8 } }, "grok-4-20-non-reasoning": { "id": "grok-4-20-non-reasoning", "name": "Grok 4.20 (Non-Reasoning)", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "glm-4.7-flashx": { "id": "glm-4.7-flashx", "name": "GLM-4.7-FlashX", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.07, "output": 0.4, "cache_read": 0.01, "cache_write": 0 } }, "qwen3-30b-a3b-instruct-2507": { "id": "qwen3-30b-a3b-instruct-2507", "name": "Qwen3 30B A3B Instruct (2507)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-08", "last_updated": "2025-07-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 8192 }, "cost": { "input": 0.1, "output": 0.3 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.931, "output": 2.93, "cache_read": 0.173, "cache_write": 0 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1050000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "qwen3-next-80b-a3b-thinking": { "id": "qwen3-next-80b-a3b-thinking", "name": "Qwen3-Next 80B-A3B (Thinking)", "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 1.2 } }, "glm-4.6": { "id": "glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.55, "output": 2.2, "cache_read": 0.11, "cache_write": 0 } }, "qwen35-397b-a17b": { "id": "qwen35-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.6, "output": 3.6 } }, "qwen3-235b-a22b-thinking-2507": { "id": "qwen3-235b-a22b-thinking-2507", "name": "Qwen3 235B A22B Thinking (2507)", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-08", "last_updated": "2025-07-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 8192 }, "cost": { "input": 0.2, "output": 0.6 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.06 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemini-pro-latest": { "id": "gemini-pro-latest", "name": "Gemini Pro Latest", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-27", "last_updated": "2026-02-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "glm-4.5-x": { "id": "glm-4.5-x", "name": "GLM-4.5 X", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "status": "beta", "cost": { "input": 2.2, "output": 8.9, "cache_read": 0.45 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "grok-4-1-fast-non-reasoning": { "id": "grok-4-1-fast-non-reasoning", "name": "Grok 4.1 Fast Non-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "grok-4-20-beta-0309-reasoning": { "id": "grok-4-20-beta-0309-reasoning", "name": "Grok 4.20 (Reasoning)", "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1024000, "output": 131072 }, "cost": { "input": 1.26, "output": 3.96, "cache_read": 0.234, "cache_write": 0 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.15, "output": 0.9, "cache_read": 0.015 } }, "glm-4.5-airx": { "id": "glm-4.5-airx", "name": "GLM-4.5 AirX", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.1, "output": 4.5, "cache_read": 0.22 } }, "gpt-5-chat-latest": { "id": "gpt-5-chat-latest", "name": "GPT-5 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "ministral-3b-2512": { "id": "ministral-3b-2512", "name": "Ministral 3B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.1, "output": 0.1 } }, "qwen2-5-vl-72b-instruct": { "id": "qwen2-5-vl-72b-instruct", "name": "Qwen2.5-VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0.13, "output": 0.4 } }, "glm-4.6v-flashx": { "id": "glm-4.6v-flashx", "name": "GLM-4.6V FlashX", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16000 }, "cost": { "input": 0.04, "output": 0.4, "cache_read": 0.004 } }, "minimax-m3": { "id": "minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 128000 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.12 } }, "qwen3-vl-plus": { "id": "qwen3-vl-plus", "name": "Qwen3-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.2, "output": 1.6, "reasoning": 4.8, "cache_read": 0.04, "cache_write": 0.25 } }, "claude-opus-4-5-20251101": { "id": "claude-opus-4-5-20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Codex GPT for repository edits, code review, and practical software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "minimax-m2": { "id": "minimax-m2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 128000 }, "cost": { "input": 0.2, "output": 1, "cache_read": 0.03 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5.3-chat-latest": { "id": "gpt-5.3-chat-latest", "name": "GPT-5.3 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "grok-4-5": { "id": "grok-4-5", "name": "Grok 4.5", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 500000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.5 } }, "claude-3-opus": { "id": "claude-3-opus", "name": "Claude 3 Opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2024-03-04", "last_updated": "2024-03-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "qwen-omni-turbo": { "id": "qwen-omni-turbo", "name": "Qwen-Omni Turbo", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01-19", "last_updated": "2025-03-26", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0.2, "output": 0.8 } }, "o3-mini": { "id": "o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "minimax-m2.7": { "id": "minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } }, "qwen-flash": { "id": "qwen-flash", "name": "Qwen Flash", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.01, "cache_write": 0.0625 } }, "qwen3-4b-fp8": { "id": "qwen3-4b-fp8", "name": "Qwen3 4B FP8", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.03, "output": 0.03 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01 } }, "qwen2-5-vl-32b-instruct": { "id": "qwen2-5-vl-32b-instruct", "name": "Qwen2.5 VL 32B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-03-15", "last_updated": "2025-03-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 1.4, "output": 4.2 } }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25 } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.405, "output": 1.98, "cache_read": 0.225 } }, "seed-1-6-250915": { "id": "seed-1-6-250915", "name": "Seed 1.6 (250915)", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.05 } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 3.125 } }, "gpt-5.2-chat-latest": { "id": "gpt-5.2-chat-latest", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "minimax-text-01": { "id": "minimax-text-01", "name": "MiniMax Text 01", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-01-15", "last_updated": "2025-01-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.2, "output": 1.1 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32766 }, "cost": { "input": 0.05, "output": 0.25 } }, "qwen-vl-max": { "id": "qwen-vl-max", "name": "Qwen-VL Max", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-08", "last_updated": "2025-08-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.8, "output": 3.2 } }, "llama-3-8b-instruct": { "id": "llama-3-8b-instruct", "name": "Llama 3 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.04, "output": 0.04 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "sonar": { "id": "sonar", "name": "Sonar", "description": "Fast web-grounded Sonar for current answers, citations, and lightweight retrieval", "family": "sonar", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 130000, "output": 4096 }, "cost": { "input": 1, "output": 1 } }, "qwen-max": { "id": "qwen-max", "name": "Qwen Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-03", "last_updated": "2025-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 1.6, "output": 6.4 } }, "o1": { "id": "o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "minimax-m2.5-highspeed": { "id": "minimax-m2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.03, "cache_write": 0.375 } }, "deepseek-v3.1": { "id": "deepseek-v3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.56, "output": 1.68, "cache_read": 0.112 } }, "ministral-14b-2512": { "id": "ministral-14b-2512", "name": "Ministral 14B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.2, "output": 0.2 } }, "sonar-pro": { "id": "sonar-pro", "name": "Sonar Pro", "description": "Deeper Sonar search model with broader retrieval and stronger synthesis", "family": "sonar-pro", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 3, "output": 15 } }, "glm-4.6v": { "id": "glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9, "cache_read": 0.05 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "claude-haiku-4-5-free": { "id": "claude-haiku-4-5-free", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 200000 }, "cost": { "input": 0, "output": 0 } }, "minimax-m2.1-lightning": { "id": "minimax-m2.1-lightning", "name": "MiniMax M2.1 Lightning", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 131072 }, "cost": { "input": 0.12, "output": 0.48 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "mimo-v2.5": { "id": "mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028, "tiers": [ { "input": 0.8, "output": 4, "cache_read": 0.16, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.8, "output": 4, "cache_read": 0.16 } } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.4, "output": 2.2, "cache_read": 0.08 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "qwen3-coder-next": { "id": "qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.108, "output": 0.675, "cache_read": 0.06 } }, "llama-4-maverick-17b-instruct": { "id": "llama-4-maverick-17b-instruct", "name": "Llama 4 Maverick 17B Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 2048 }, "cost": { "input": 0.27, "output": 0.85 } }, "qwen3-coder-480b-a35b-instruct": { "id": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3-Coder 480B-A35B Instruct", "description": "Open Qwen coding heavyweight for repository reasoning and agentic engineering", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 1.3 } }, "devstral-2512": { "id": "devstral-2512", "name": "Devstral 2", "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemma-4-26b-a4b-it": { "id": "gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.07, "output": 0.34 } }, "qwen3.5-9b": { "id": "qwen3.5-9b", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.1, "output": 0.15 } }, "qwq-plus": { "id": "qwq-plus", "name": "QwQ Plus", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-03-05", "last_updated": "2025-03-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.8, "output": 2.4 } }, "grok-4-1-fast-reasoning": { "id": "grok-4-1-fast-reasoning", "name": "Grok 4.1 Fast Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "qwen3-vl-30b-a3b-instruct": { "id": "qwen3-vl-30b-a3b-instruct", "name": "Qwen3 VL 30B A3B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-02", "last_updated": "2025-10-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.2, "output": 0.7 } }, "qwen3-vl-flash": { "id": "qwen3-vl-flash", "name": "Qwen3 VL Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-09", "last_updated": "2025-10-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.01 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "llama-3.1-nemotron-ultra-253b": { "id": "llama-3.1-nemotron-ultra-253b", "name": "Llama 3.1 Nemotron Ultra 253B", "description": "Flagship Nemotron model for high-throughput reasoning and complex agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-04-07", "last_updated": "2025-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.6, "output": 1.8 } }, "gpt-4-turbo": { "id": "gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "qwen3-coder-flash": { "id": "qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.5, "cache_read": 0.06, "cache_write": 0.375 } }, "grok-build-0-1": { "id": "grok-build-0-1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 4, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2, "output": 4, "cache_read": 0.4 } } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "grok-4": { "id": "grok-4", "name": "Grok 4", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75 } }, "llama-3.2-11b-instruct": { "id": "llama-3.2-11b-instruct", "name": "Llama 3.2 11B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.07, "output": 0.33 } }, "gpt-5.4-pro": { "id": "gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 98304 }, "cost": { "input": 0.13, "output": 0.85, "cache_read": 0.025, "cache_write": 0 } }, "kimi-k2.7-code-highspeed": { "id": "kimi-k2.7-code-highspeed", "name": "Kimi K2.7 Code Highspeed", "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.9, "output": 8, "cache_read": 0.38 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05 } }, "glm-4.7-flash": { "id": "glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.06, "output": 0.4, "cache_read": 0.01, "cache_write": 0 } }, "qwen3-vl-235b-a22b-instruct": { "id": "qwen3-vl-235b-a22b-instruct", "name": "Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.3, "output": 1.5 } }, "gpt-4o-search-preview": { "id": "gpt-4o-search-preview", "name": "GPT-4o Search Preview", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "llama-3.2-3b-instruct": { "id": "llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32000 }, "cost": { "input": 0.03, "output": 0.05 } }, "mimo-v2.5-pro": { "id": "mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "gpt-5.5-pro": { "id": "gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "gpt-4o-mini": { "id": "gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32766 }, "cost": { "input": 0.04, "output": 0.15 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 203000, "output": 131072 }, "cost": { "input": 0.72, "output": 2.3, "cache_read": 0.144, "cache_write": 0 } }, "qwen-max-latest": { "id": "qwen-max-latest", "name": "Qwen Max Latest", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-01-25", "last_updated": "2025-01-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 1.6, "output": 6.4 } }, "mistral-large-2512": { "id": "mistral-large-2512", "name": "Mistral Large 3", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.5, "output": 1.5 } }, "qwen-turbo": { "id": "qwen-turbo", "name": "Qwen Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-11-01", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.05, "output": 0.2, "reasoning": 0.5 } }, "custom": { "id": "custom", "name": "Custom Model", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "auto", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25 } }, "grok-4-20-reasoning": { "id": "grok-4-20-reasoning", "name": "Grok 4.20 (Reasoning)", "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 16384 }, "cost": { "input": 0.26, "output": 0.38, "cache_read": 0.13 } }, "qwen3.6-max-preview": { "id": "qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.3, "output": 7.8, "cache_read": 0.13, "cache_write": 1.625 } }, "auto": { "id": "auto", "name": "Auto Route", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "auto", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "codestral-2508": { "id": "codestral-2508", "name": "Codestral", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.3, "output": 0.9 } }, "qwen3-235b-a22b-fp8": { "id": "qwen3-235b-a22b-fp8", "name": "Qwen3 235B A22B FP8", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 8192 }, "cost": { "input": 0.2, "output": 0.8 } }, "fugu-ultra": { "id": "fugu-ultra", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-06-22", "last_updated": "2026-06-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "qwen-vl-plus": { "id": "qwen-vl-plus", "name": "Qwen-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-08-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.21, "output": 0.64 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "alibaba-coding-plan-cn": { "id": "alibaba-coding-plan-cn", "env": ["ALIBABA_CODING_PLAN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://coding.dashscope.aliyuncs.com/v1", "name": "Alibaba Coding Plan (China)", "doc": "https://help.aliyun.com/zh/model-studio/coding-plan", "models": { "qwen3-coder-plus": { "id": "qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.5, "cache_write": 3.125 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1875, "output": 1.125, "cache_write": 0.234375 } }, "qwen3-max-2026-01-23": { "id": "qwen3-max-2026-01-23", "name": "Qwen3 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-23", "last_updated": "2026-01-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 24576 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3-coder-next": { "id": "qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "abacus": { "id": "abacus", "env": ["ABACUS_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://routellm.abacus.ai/v1", "name": "Abacus", "doc": "https://abacus.ai/help/api", "models": { "o3": { "id": "o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "gemini-3.1-flash-lite": { "id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025 } }, "route-llm": { "id": "route-llm", "name": "RouteLLM", "description": "RouteLLM routes prompts to an appropriate Abacus-backed text-generation model", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-01-01", "last_updated": "2026-07-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "grok-code-fast-1": { "id": "grok-code-fast-1", "name": "Grok Code Fast 1", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.2, "output": 1.5 } }, "gpt-5.3-codex-xhigh": { "id": "gpt-5.3-codex-xhigh", "name": "GPT-5.3 Codex XHigh", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "llama-3.3-70b-versatile": { "id": "llama-3.3-70b-versatile", "name": "Llama 3.3 70B Versatile", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.59, "output": 0.79 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5 } }, "grok-4.3": { "id": "grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03 } }, "gpt-4o": { "id": "gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "o4-mini": { "id": "o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4 } }, "qwen3-max": { "id": "qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 1.2, "output": 6 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 } }, "o3-pro": { "id": "o3-pro", "name": "o3-pro", "description": "High-effort o3 tier for difficult technical reasoning and careful answers", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 20, "output": 40 } }, "muse-spark-1.1": { "id": "muse-spark-1.1", "name": "Muse Spark 1.1", "description": "Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.", "family": "muse", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 1.25, "output": 4.25, "cache_read": 0.15 } }, "claude-opus-4-1-20250805": { "id": "claude-opus-4-1-20250805", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "grok-4-1-fast-non-reasoning": { "id": "grok-4-1-fast-non-reasoning", "name": "Grok 4.1 Fast (Non-Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-11-17", "last_updated": "2025-11-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 16384 }, "cost": { "input": 0.2, "output": 0.5 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "claude-opus-4-5-20251101": { "id": "claude-opus-4-5-20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Codex GPT for repository edits, code review, and practical software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.1-codex-max": { "id": "gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "claude-opus-4-20250514": { "id": "claude-opus-4-20250514", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-05-14", "last_updated": "2025-05-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75 } }, "gpt-5.3-chat-latest": { "id": "gpt-5.3-chat-latest", "name": "GPT-5.3 Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "gemini-3-pro-image-preview": { "id": "gemini-3-pro-image-preview", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 32768 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "gemini-2.5-flash-image": { "id": "gemini-2.5-flash-image", "name": "Nano Banana", "description": "Nano Banana image model for fast generation, edits, and character-consistent assets", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.3, "output": 30 } }, "o3-mini": { "id": "o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "kimi-k2-turbo-preview": { "id": "kimi-k2-turbo-preview", "name": "Kimi K2 Turbo Preview", "description": "Fast Kimi model for responsive chat, coding help, and agent loops", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-08", "last_updated": "2025-07-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.15, "output": 8 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.18 } }, "grok-4-0709": { "id": "grok-4-0709", "name": "Grok 4", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 3, "output": 15 } }, "grok-4.5": { "id": "grok-4.5", "name": "Grok 4.5", "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 32768 }, "cost": { "input": 2, "output": 6 } }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1, "output": 6, "cache_read": 0.1 } }, "claude-sonnet-4-20250514": { "id": "claude-sonnet-4-20250514", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-05-14", "last_updated": "2025-05-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.6, "output": 3 } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "gpt-5.1-chat-latest": { "id": "gpt-5.1-chat-latest", "name": "GPT-5.1 Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "gpt-5.2-chat-latest": { "id": "gpt-5.2-chat-latest", "name": "GPT-5.2 Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2026-01-01", "last_updated": "2026-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50 } }, "gpt-4o-2024-11-20": { "id": "gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "claude-3-7-sonnet-20250219": { "id": "claude-3-7-sonnet-20250219", "name": "Claude Sonnet 3.7", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10-31", "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "gemini-3.1-flash-image-preview": { "id": "gemini-3.1-flash-image-preview", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 1048576, "output": 32768 }, "cost": { "input": 0.5, "output": 3 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "qwen-2.5-coder-32b": { "id": "qwen-2.5-coder-32b", "name": "Qwen 2.5 Coder 32B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-11-11", "last_updated": "2024-11-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.79, "output": 0.79 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "grok-4-fast-non-reasoning": { "id": "grok-4-fast-non-reasoning", "name": "Grok 4 Fast (Non-Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 16384 }, "cost": { "input": 0.2, "output": 0.5 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05 } }, "mimo-v2-pro": { "id": "mimo-v2-pro", "name": "MiMo-V2-Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2 } }, "gpt-4o-mini": { "id": "gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "gpt-5-codex": { "id": "gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gemini-3.1-flash-lite-preview": { "id": "gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "cache_write": 1 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "meta-llama/Meta-Llama-3.3-70B-Instruct": { "id": "meta-llama/Meta-Llama-3.3-70B-Instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.59, "output": 0.79 } }, "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "name": "Llama 4 Maverick 17B Instruct", "description": "Open multimodal Llama for strong reasoning with efficient everyday serving", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 8192 }, "cost": { "input": 0.14, "output": 0.59 } }, "meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": { "id": "meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo", "name": "Llama 3.1 405B Instruct Turbo", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 3.5, "output": 3.5 } }, "meta-llama/Meta-Llama-3.1-8B-Instruct": { "id": "meta-llama/Meta-Llama-3.1-8B-Instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.02, "output": 0.05 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.14, "output": 0.4 } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.32, "output": 3.2 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen3-Coder 480B-A35B Instruct", "description": "Open Qwen coding heavyweight for repository reasoning and agentic engineering", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.29, "output": 1.2 } }, "Qwen/QwQ-32B": { "id": "Qwen/QwQ-32B", "name": "QwQ 32B", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2024-11-28", "last_updated": "2024-11-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.4, "output": 0.4 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B A22B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0.13, "output": 0.6 } }, "Qwen/Qwen3-32B": { "id": "Qwen/Qwen3-32B", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.09, "output": 0.29 } }, "Qwen/Qwen2.5-72B-Instruct": { "id": "Qwen/Qwen2.5-72B-Instruct", "name": "Qwen 2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.11, "output": 0.38 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.08, "output": 0.44 } }, "zai-org/GLM-4.6": { "id": "zai-org/GLM-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1, "output": 3.2 } }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "zai-org/GLM-4.5": { "id": "zai-org/GLM-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 96000 }, "cost": { "input": 0.6, "output": 2.2 } }, "deepseek-ai/DeepSeek-R1": { "id": "deepseek-ai/DeepSeek-R1", "name": "DeepSeek R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 3, "output": 7 } }, "deepseek-ai/DeepSeek-V3.1-Terminus": { "id": "deepseek-ai/DeepSeek-V3.1-Terminus", "name": "DeepSeek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.27, "output": 1 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.03 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.15 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-15", "last_updated": "2025-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.27, "output": 0.4 } }, "MiniMaxAI/MiniMax-M3": { "id": "MiniMaxAI/MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMaxAI/MiniMax-M2.7": { "id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "deepseek/deepseek-v3.1": { "id": "deepseek/deepseek-v3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.55, "output": 1.66 } } } }, "cloudferro-sherlock": { "id": "cloudferro-sherlock", "env": ["CLOUDFERRO_SHERLOCK_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api-sherlock.cloudferro.com/openai/v1/", "name": "CloudFerro Sherlock", "doc": "https://docs.sherlock.cloudferro.com/", "models": { "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10-09", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 70000, "output": 70000 }, "cost": { "input": 2.92, "output": 2.92 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "OpenAI GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 2.92, "output": 2.92 } }, "speakleash/Bielik-11B-v3.0-Instruct": { "id": "speakleash/Bielik-11B-v3.0-Instruct", "name": "Bielik 11B v3.0 Instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.67, "output": 0.67 } }, "speakleash/Bielik-11B-v2.6-Instruct": { "id": "speakleash/Bielik-11B-v2.6-Instruct", "name": "Bielik 11B v2.6 Instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.67, "output": 0.67 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196000, "input": 180000, "output": 16000 }, "cost": { "input": 0.3, "output": 1.2 } } } }, "ollama-cloud": { "id": "ollama-cloud", "env": ["OLLAMA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://ollama.com/v1", "name": "Ollama Cloud", "doc": "https://docs.ollama.com/cloud", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "deepseek-v4-flash", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 1048576 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "minimax-m2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "knowledge": "2025-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 } }, "devstral-small-2:24b": { "id": "devstral-small-2:24b", "name": "devstral-small-2:24b", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2025-12-09", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "glm-4.7": { "id": "glm-4.7", "name": "glm-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2025-12-22", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 } }, "cogito-2.1:671b": { "id": "cogito-2.1:671b", "name": "cogito-2.1:671b", "description": "Legacy model retained for compatibility with older integrations", "family": "cogito", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2025-11-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32000 }, "status": "deprecated" }, "minimax-m2.1": { "id": "minimax-m2.1", "name": "minimax-m2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "release_date": "2025-12-23", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 } }, "gpt-oss:120b": { "id": "gpt-oss:120b", "name": "gpt-oss:120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "release_date": "2025-08-05", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 } }, "nemotron-3-nano:30b": { "id": "nemotron-3-nano:30b", "name": "nemotron-3-nano:30b", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-12-15", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 } }, "ministral-3:8b": { "id": "ministral-3:8b", "name": "ministral-3:8b", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2024-12-01", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 128000 } }, "rnj-1:8b": { "id": "rnj-1:8b", "name": "rnj-1:8b", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "rnj", "attachment": false, "reasoning": false, "tool_call": true, "release_date": "2025-12-06", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 4096 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "kimi-k2.7-code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "glm-5.1": { "id": "glm-5.1", "name": "glm-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "release_date": "2026-03-27", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "deepseek-v4-pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 1048576 } }, "glm-4.6": { "id": "glm-4.6", "name": "glm-4.6", "description": "Legacy model retained for compatibility with older integrations", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2025-09-29", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "status": "deprecated" }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "kimi-k2-thinking", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated" }, "nemotron-3-super": { "id": "nemotron-3-super", "name": "nemotron-3-super", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 } }, "ministral-3:14b": { "id": "ministral-3:14b", "name": "ministral-3:14b", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2024-12-01", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 128000 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 976000, "output": 131072 } }, "minimax-m3": { "id": "minimax-m3", "name": "minimax-m3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax-m3", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-31", "last_updated": "2026-05-31", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 131072 } }, "minimax-m2": { "id": "minimax-m2", "name": "minimax-m2", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "release_date": "2025-10-23", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128000 }, "status": "deprecated" }, "qwen3-next:80b": { "id": "qwen3-next:80b", "name": "qwen3-next:80b", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "release_date": "2025-09-15", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "status": "deprecated" }, "qwen3-coder:480b": { "id": "qwen3-coder:480b", "name": "qwen3-coder:480b", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "release_date": "2025-07-22", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 } }, "kimi-k2:1t": { "id": "kimi-k2:1t", "name": "kimi-k2:1t", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "knowledge": "2024-10", "release_date": "2025-07-11", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated" }, "minimax-m2.7": { "id": "minimax-m2.7", "name": "minimax-m2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 } }, "deepseek-v3.1:671b": { "id": "deepseek-v3.1:671b", "name": "deepseek-v3.1:671b", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2025-08-21", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "kimi-k2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "gemma4:31b": { "id": "gemma4:31b", "name": "gemma4:31b", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "ministral-3:3b": { "id": "ministral-3:3b", "name": "ministral-3:3b", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2024-10-22", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 128000 } }, "gemma3:12b": { "id": "gemma3:12b", "name": "gemma3:12b", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "release_date": "2024-12-01", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 } }, "gemma3:4b": { "id": "gemma3:4b", "name": "gemma3:4b", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "release_date": "2024-12-01", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 } }, "qwen3.5:397b": { "id": "qwen3.5:397b", "name": "qwen3.5:397b", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "release_date": "2026-02-15", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "kimi-k2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "qwen3-coder-next": { "id": "qwen3-coder-next", "name": "qwen3-coder-next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "release_date": "2026-02-02", "last_updated": "2026-02-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 } }, "qwen3-vl:235b-instruct": { "id": "qwen3-vl:235b-instruct", "name": "qwen3-vl:235b-instruct", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2025-09-22", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "status": "deprecated" }, "mistral-large-3:675b": { "id": "mistral-large-3:675b", "name": "mistral-large-3:675b", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2025-12-02", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } }, "gemma3:27b": { "id": "gemma3:27b", "name": "gemma3:27b", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "release_date": "2025-07-27", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 } }, "qwen3-vl:235b": { "id": "qwen3-vl:235b", "name": "qwen3-vl:235b", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "release_date": "2025-09-22", "last_updated": "2026-01-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "status": "deprecated" }, "nemotron-3-ultra": { "id": "nemotron-3-ultra", "name": "nemotron-3-ultra", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 128000 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "gemini-3-flash-preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2026-04-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 65536 } }, "gpt-oss:20b": { "id": "gpt-oss:20b", "name": "gpt-oss:20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "release_date": "2025-08-05", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 } }, "glm-5": { "id": "glm-5", "name": "glm-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "deepseek-v3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "release_date": "2025-06-15", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 } }, "devstral-2:123b": { "id": "devstral-2:123b", "name": "devstral-2:123b", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "release_date": "2025-12-09", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 } } } }, "cloudflare-ai-gateway": { "id": "cloudflare-ai-gateway", "env": ["CLOUDFLARE_API_TOKEN", "CLOUDFLARE_ACCOUNT_ID", "CLOUDFLARE_GATEWAY_ID"], "npm": "ai-gateway-provider", "name": "Cloudflare AI Gateway", "doc": "https://developers.cloudflare.com/ai-gateway/", "models": { "workers-ai/@cf/baai/bge-m3": { "id": "workers-ai/@cf/baai/bge-m3", "name": "BGE M3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.012, "output": 0 } }, "workers-ai/@cf/baai/bge-small-en-v1.5": { "id": "workers-ai/@cf/baai/bge-small-en-v1.5", "name": "BGE Small EN v1.5", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.02, "output": 0 } }, "workers-ai/@cf/baai/bge-reranker-base": { "id": "workers-ai/@cf/baai/bge-reranker-base", "name": "BGE Reranker Base", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-09", "last_updated": "2025-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.0031, "output": 0 } }, "workers-ai/@cf/baai/bge-base-en-v1.5": { "id": "workers-ai/@cf/baai/bge-base-en-v1.5", "name": "BGE Base EN v1.5", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.067, "output": 0 } }, "workers-ai/@cf/baai/bge-large-en-v1.5": { "id": "workers-ai/@cf/baai/bge-large-en-v1.5", "name": "BGE Large EN v1.5", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.2, "output": 0 } }, "workers-ai/@cf/ai4bharat/indictrans2-en-indic-1B": { "id": "workers-ai/@cf/ai4bharat/indictrans2-en-indic-1B", "name": "IndicTrans2 EN-Indic 1B", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "indictrans", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.34, "output": 0.34 } }, "workers-ai/@cf/ibm-granite/granite-4.0-h-micro": { "id": "workers-ai/@cf/ibm-granite/granite-4.0-h-micro", "name": "IBM Granite 4.0 H Micro", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "granite", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.017, "output": 0.11 } }, "workers-ai/@cf/huggingface/distilbert-sst-2-int8": { "id": "workers-ai/@cf/huggingface/distilbert-sst-2-int8", "name": "DistilBERT SST-2 INT8", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "distilbert", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.026, "output": 0 } }, "workers-ai/@cf/moonshotai/kimi-k2.5": { "id": "workers-ai/@cf/moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "workers-ai/@cf/moonshotai/kimi-k2.6": { "id": "workers-ai/@cf/moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "workers-ai/@cf/mistral/mistral-7b-instruct-v0.1": { "id": "workers-ai/@cf/mistral/mistral-7b-instruct-v0.1", "name": "Mistral 7B Instruct v0.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.11, "output": 0.19 } }, "workers-ai/@cf/google/gemma-3-12b-it": { "id": "workers-ai/@cf/google/gemma-3-12b-it", "name": "Gemma 3 12B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-11", "last_updated": "2025-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.35, "output": 0.56 } }, "workers-ai/@cf/myshell-ai/melotts": { "id": "workers-ai/@cf/myshell-ai/melotts", "name": "MyShell MeloTTS", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "melotts", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "workers-ai/@cf/openai/gpt-oss-120b": { "id": "workers-ai/@cf/openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.35, "output": 0.75 } }, "workers-ai/@cf/openai/gpt-oss-20b": { "id": "workers-ai/@cf/openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.2, "output": 0.3 } }, "workers-ai/@cf/mistralai/mistral-small-3.1-24b-instruct": { "id": "workers-ai/@cf/mistralai/mistral-small-3.1-24b-instruct", "name": "Mistral Small 3.1 24B Instruct", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-11", "last_updated": "2025-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.35, "output": 0.56 } }, "workers-ai/@cf/nvidia/nemotron-3-120b-a12b": { "id": "workers-ai/@cf/nvidia/nemotron-3-120b-a12b", "name": "Nemotron 3 Super 120B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.5, "output": 1.5 } }, "workers-ai/@cf/pfnet/plamo-embedding-1b": { "id": "workers-ai/@cf/pfnet/plamo-embedding-1b", "name": "PLaMo Embedding 1B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "plamo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.019, "output": 0 } }, "workers-ai/@cf/deepgram/aura-2-es": { "id": "workers-ai/@cf/deepgram/aura-2-es", "name": "Deepgram Aura 2 (ES)", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "aura", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "workers-ai/@cf/deepgram/aura-2-en": { "id": "workers-ai/@cf/deepgram/aura-2-en", "name": "Deepgram Aura 2 (EN)", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "aura", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "workers-ai/@cf/deepgram/nova-3": { "id": "workers-ai/@cf/deepgram/nova-3", "name": "Deepgram Nova 3", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "nova", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "workers-ai/@cf/zai-org/glm-4.7-flash": { "id": "workers-ai/@cf/zai-org/glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.06, "output": 0.4 } }, "workers-ai/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b": { "id": "workers-ai/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b", "name": "DeepSeek R1 Distill Qwen 32B", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "deepseek-thinking", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.5, "output": 4.88 } }, "workers-ai/@cf/qwen/qwen3-embedding-0.6b": { "id": "workers-ai/@cf/qwen/qwen3-embedding-0.6b", "name": "Qwen3 Embedding 0.6B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.012, "output": 0 } }, "workers-ai/@cf/qwen/qwen3-30b-a3b-fp8": { "id": "workers-ai/@cf/qwen/qwen3-30b-a3b-fp8", "name": "Qwen3 30B A3B FP8", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.051, "output": 0.34 } }, "workers-ai/@cf/qwen/qwen2.5-coder-32b-instruct": { "id": "workers-ai/@cf/qwen/qwen2.5-coder-32b-instruct", "name": "Qwen 2.5 Coder 32B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-11", "last_updated": "2025-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.66, "output": 1 } }, "workers-ai/@cf/qwen/qwq-32b": { "id": "workers-ai/@cf/qwen/qwq-32b", "name": "QwQ 32B", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-11", "last_updated": "2025-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.66, "output": 1 } }, "workers-ai/@cf/pipecat-ai/smart-turn-v2": { "id": "workers-ai/@cf/pipecat-ai/smart-turn-v2", "name": "Pipecat Smart Turn v2", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "smart-turn", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "workers-ai/@cf/meta/llama-3.1-8b-instruct": { "id": "workers-ai/@cf/meta/llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.28, "output": 0.8299999999999998 } }, "workers-ai/@cf/meta/m2m100-1.2b": { "id": "workers-ai/@cf/meta/m2m100-1.2b", "name": "M2M100 1.2B", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "m2m", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.34, "output": 0.34 } }, "workers-ai/@cf/meta/llama-3.2-1b-instruct": { "id": "workers-ai/@cf/meta/llama-3.2-1b-instruct", "name": "Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.027, "output": 0.2 } }, "workers-ai/@cf/meta/llama-3.2-11b-vision-instruct": { "id": "workers-ai/@cf/meta/llama-3.2-11b-vision-instruct", "name": "Llama 3.2 11B Vision Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.049, "output": 0.68 } }, "workers-ai/@cf/meta/llama-4-scout-17b-16e-instruct": { "id": "workers-ai/@cf/meta/llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout 17B 16E Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.27, "output": 0.85 } }, "workers-ai/@cf/meta/llama-guard-3-8b": { "id": "workers-ai/@cf/meta/llama-guard-3-8b", "name": "Llama Guard 3 8B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.48, "output": 0.03 } }, "workers-ai/@cf/meta/llama-3-8b-instruct-awq": { "id": "workers-ai/@cf/meta/llama-3-8b-instruct-awq", "name": "Llama 3 8B Instruct AWQ", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.12, "output": 0.27 } }, "workers-ai/@cf/meta/llama-3.1-8b-instruct-awq": { "id": "workers-ai/@cf/meta/llama-3.1-8b-instruct-awq", "name": "Llama 3.1 8B Instruct AWQ", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.12, "output": 0.27 } }, "workers-ai/@cf/meta/llama-3.3-70b-instruct-fp8-fast": { "id": "workers-ai/@cf/meta/llama-3.3-70b-instruct-fp8-fast", "name": "Llama 3.3 70B Instruct FP8 Fast", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.29, "output": 2.25 } }, "workers-ai/@cf/meta/llama-3-8b-instruct": { "id": "workers-ai/@cf/meta/llama-3-8b-instruct", "name": "Llama 3 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.28, "output": 0.83 } }, "workers-ai/@cf/meta/llama-3.1-8b-instruct-fp8": { "id": "workers-ai/@cf/meta/llama-3.1-8b-instruct-fp8", "name": "Llama 3.1 8B Instruct FP8", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.29 } }, "workers-ai/@cf/meta/llama-2-7b-chat-fp16": { "id": "workers-ai/@cf/meta/llama-2-7b-chat-fp16", "name": "Llama 2 7B Chat FP16", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.56, "output": 6.67 } }, "workers-ai/@cf/meta/llama-3.2-3b-instruct": { "id": "workers-ai/@cf/meta/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.051, "output": 0.34 } }, "workers-ai/@cf/facebook/bart-large-cnn": { "id": "workers-ai/@cf/facebook/bart-large-cnn", "name": "BART Large CNN", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "bart", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-09", "last_updated": "2025-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "workers-ai/@cf/aisingapore/gemma-sea-lion-v4-27b-it": { "id": "workers-ai/@cf/aisingapore/gemma-sea-lion-v4-27b-it", "name": "Gemma SEA-LION v4 27B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.35, "output": 0.56 } }, "openai/o3": { "id": "openai/o3", "name": "o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-3.5-turbo": { "id": "openai/gpt-3.5-turbo", "name": "GPT-3.5-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2021-09-01", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 1.25 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/gpt-4": { "id": "openai/gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 30, "output": 60 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.28 } }, "openai/o3-pro": { "id": "openai/o3-pro", "name": "o3-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 20, "output": 80 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "o3-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "ai-gateway-provider" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/o1": { "id": "openai/o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "ai-gateway-provider" }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "openai/gpt-4-turbo": { "id": "openai/gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.08 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "ai-gateway-provider" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "anthropic/claude-3.5-haiku": { "id": "anthropic/claude-3.5-haiku", "name": "Claude Haiku 3.5 (latest)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "anthropic/claude-3.5-sonnet": { "id": "anthropic/claude-3.5-sonnet", "name": "Claude Sonnet 3.5 v2", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4-5": { "id": "anthropic/claude-opus-4-5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4-5": { "id": "anthropic/claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4-7": { "id": "anthropic/claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-5": { "id": "anthropic/claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "anthropic/claude-3-sonnet": { "id": "anthropic/claude-3-sonnet", "name": "Claude Sonnet 3", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-03-04", "last_updated": "2024-03-04", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 0.3 } }, "anthropic/claude-opus-4-8": { "id": "anthropic/claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-3-opus": { "id": "anthropic/claude-3-opus", "name": "Claude Opus 3", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-02-29", "last_updated": "2024-02-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-3-5-haiku": { "id": "anthropic/claude-3-5-haiku", "name": "Claude Haiku 3.5 (latest)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "anthropic/claude-opus-4-1": { "id": "anthropic/claude-opus-4-1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-fable-5": { "id": "anthropic/claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "anthropic/claude-haiku-4-5": { "id": "anthropic/claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-opus-4-6": { "id": "anthropic/claude-opus-4-6", "name": "Claude Opus 4.6 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "anthropic/claude-3-haiku": { "id": "anthropic/claude-3-haiku", "name": "Claude Haiku 3", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-03-13", "last_updated": "2024-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.03, "cache_write": 0.3 } }, "anthropic/claude-sonnet-4-6": { "id": "anthropic/claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "ai-gateway-provider" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Claude Opus 4 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } } } }, "moonshotai-cn": { "id": "moonshotai-cn", "env": ["MOONSHOT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.moonshot.cn/v1", "name": "Moonshot AI (China)", "doc": "https://platform.moonshot.cn/docs/api/chat", "models": { "kimi-k2.7-code-highspeed": { "id": "kimi-k2.7-code-highspeed", "name": "Kimi K2.7 Code HighSpeed", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.9, "output": 8, "cache_read": 0.38 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "kimi-k2-turbo-preview": { "id": "kimi-k2-turbo-preview", "name": "Kimi K2 Turbo", "description": "Fast Kimi model for responsive chat, coding help, and agent loops", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 2.4, "output": 10, "cache_read": 0.6 } }, "kimi-k2-0711-preview": { "id": "kimi-k2-0711-preview", "name": "Kimi K2 0711", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-07-14", "last_updated": "2025-07-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "kimi-k2-thinking-turbo": { "id": "kimi-k2-thinking-turbo", "name": "Kimi K2 Thinking Turbo", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.15, "output": 8, "cache_read": 0.15 } }, "kimi-k2-0905-preview": { "id": "kimi-k2-0905-preview", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } } } }, "morph": { "id": "morph", "env": ["MORPH_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.morphllm.com/v1", "name": "Morph", "doc": "https://docs.morphllm.com/api-reference/introduction", "models": { "morph-v3-fast": { "id": "morph-v3-fast", "name": "Morph v3 Fast", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-08-15", "last_updated": "2024-08-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16000, "output": 16000 }, "cost": { "input": 0.8, "output": 1.2 } }, "morph-v3-large": { "id": "morph-v3-large", "name": "Morph v3 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-08-15", "last_updated": "2024-08-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.9, "output": 1.9 } }, "auto": { "id": "auto", "name": "Auto", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "auto", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.85, "output": 1.55 } } } }, "sakana": { "id": "sakana", "env": ["SAKANA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.sakana.ai/v1", "name": "Sakana AI", "doc": "https://console.sakana.ai/models", "models": { "fugu-ultra-20260615": { "id": "fugu-ultra-20260615", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "family": "fugu", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-15", "last_updated": "2026-06-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "provider": { "shape": "responses" }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "fugu": { "id": "fugu", "name": "Fugu", "description": "Multi-agent model for routing expert agents across complex analytical tasks", "family": "fugu", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-15", "last_updated": "2026-06-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "provider": { "shape": "responses" } }, "fugu-ultra": { "id": "fugu-ultra", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "family": "fugu", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-15", "last_updated": "2026-06-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "provider": { "shape": "responses" }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "deepinfra": { "id": "deepinfra", "env": ["DEEPINFRA_API_KEY"], "npm": "@ai-sdk/deepinfra", "name": "Deep Infra", "doc": "https://deepinfra.com/models", "models": { "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "name": "Llama 4 Maverick 17B FP8", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 16384 }, "cost": { "input": 0.2, "output": 0.8 } }, "meta-llama/Llama-4-Scout-17B-16E-Instruct": { "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct", "name": "Llama 4 Scout 17B", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 327680, "output": 16384 }, "cost": { "input": 0.1, "output": 0.3 } }, "meta-llama/Llama-3.3-70B-Instruct-Turbo": { "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo", "name": "Llama 3.3 70B Turbo", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.1, "output": 0.32 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.75, "output": 3.5, "cache_read": 0.15 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.45, "output": 2.25, "cache_read": 0.07 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.74, "output": 3.5, "cache_read": 0.15 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.13, "output": 0.38 } }, "google/gemma-4-26B-A4B-it": { "id": "google/gemma-4-26B-A4B-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.07, "output": 0.34 } }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 81920 }, "cost": { "input": 0.15, "output": 0.95 } }, "Qwen/Qwen3.7-Max": { "id": "Qwen/Qwen3.7-Max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.5, "tiers": [ { "input": 5, "output": 15, "cache_read": 1, "tier": { "type": "context", "size": 32000 } }, { "input": 6.25, "output": 18.5, "cache_read": 1.25, "tier": { "type": "context", "size": 128000 } } ] } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.32, "output": 3.2 } }, "Qwen/Qwen3-Next-80B-A3B-Instruct": { "id": "Qwen/Qwen3-Next-80B-A3B-Instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.09, "output": 1.1 } }, "Qwen/Qwen3-Max": { "id": "Qwen/Qwen3-Max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 1.2, "output": 6, "cache_read": 0.24, "tiers": [ { "input": 2.4, "output": 12, "cache_read": 0.48, "tier": { "type": "context", "size": 32000 } }, { "input": 3, "output": 15, "cache_read": 0.6, "tier": { "type": "context", "size": 128000 } } ] } }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen 3.5 397B A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-01", "last_updated": "2026-04-20", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 81920 }, "cost": { "input": 0.45, "output": 3, "cache_read": 0.22 } }, "Qwen/Qwen3.5-122B-A10B": { "id": "Qwen/Qwen3.5-122B-A10B", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.29, "output": 2.4 } }, "Qwen/Qwen3.5-27B": { "id": "Qwen/Qwen3.5-27B", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.26, "output": 2.6 } }, "Qwen/Qwen3.5-9B": { "id": "Qwen/Qwen3.5-9B", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.1, "output": 0.15 } }, "Qwen/Qwen3-32B": { "id": "Qwen/Qwen3-32B", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 16384 }, "cost": { "input": 0.08, "output": 0.28 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo", "name": "Qwen3 Coder 480B A35B Instruct Turbo", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 66536 }, "cost": { "input": 0.3, "output": 1, "cache_read": 0.1 } }, "Qwen/Qwen3.5-35B-A3B": { "id": "Qwen/Qwen3.5-35B-A3B", "name": "Qwen 3.5 35B A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-01", "last_updated": "2026-04-20", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 81920 }, "cost": { "input": 0.14, "output": 1, "cache_read": 0.05 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.037, "output": 0.17 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.03, "output": 0.14 } }, "XiaomiMiMo/MiMo-V2.5": { "id": "XiaomiMiMo/MiMo-V2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.08 } }, "XiaomiMiMo/MiMo-V2.5-Pro": { "id": "XiaomiMiMo/MiMo-V2.5-Pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 16384 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2 } }, "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": { "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning", "name": "Nemotron 3 Nano Omni 30B A3B Reasoning", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.2, "output": 0.8 } }, "nvidia/Nemotron-3-Nano-30B-A3B": { "id": "nvidia/Nemotron-3-Nano-30B-A3B", "name": "Nemotron 3 Nano 30B A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.05, "output": 0.2 } }, "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": { "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5", "name": "Llama 3.3 Nemotron Super 49B v1.5", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.4, "output": 0.4 } }, "zai-org/GLM-4.7-Flash": { "id": "zai-org/GLM-4.7-Flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0.06, "output": 0.4, "cache_read": 0.01 } }, "zai-org/GLM-4.6": { "id": "zai-org/GLM-4.6", "name": "GLM-4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.5, "output": 2, "cache_read": 0.1 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0.6, "output": 2.08, "cache_read": 0.12 } }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0.4, "output": 1.75, "cache_read": 0.08 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 32768 }, "cost": { "input": 0.93, "output": 3, "cache_read": 0.18 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 1.05, "output": 3.5, "cache_read": 0.205 } }, "deepseek-ai/DeepSeek-R1-0528": { "id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek-R1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 64000 }, "cost": { "input": 0.5, "output": 2.15, "cache_read": 0.35 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 16384 }, "cost": { "input": 0.09, "output": 0.18, "cache_read": 0.018 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 16384 }, "cost": { "input": 1.3, "output": 2.6, "cache_read": 0.1 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 64000 }, "cost": { "input": 0.26, "output": 0.38, "cache_read": 0.13 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-06", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 131072 }, "cost": { "input": 0.15, "output": 1.15, "cache_read": 0.03 } }, "MiniMaxAI/MiniMax-M3": { "id": "MiniMaxAI/MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "MiniMaxAI/MiniMax-M2.7": { "id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 131072 }, "cost": { "input": 0.25, "output": 1, "cache_read": 0.05 } } } }, "google-vertex-anthropic": { "id": "google-vertex-anthropic", "env": ["GOOGLE_VERTEX_PROJECT", "GOOGLE_VERTEX_LOCATION", "GOOGLE_APPLICATION_CREDENTIALS"], "npm": "@ai-sdk/google-vertex/anthropic", "name": "Vertex (Anthropic)", "doc": "https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude", "models": { "claude-haiku-4-5@20251001": { "id": "claude-haiku-4-5@20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "claude-opus-4@20250514": { "id": "claude-opus-4@20250514", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "claude-opus-4-1@20250805": { "id": "claude-opus-4-1@20250805", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "claude-opus-4-5@20251101": { "id": "claude-opus-4-5@20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-3-5-haiku@20241022": { "id": "claude-3-5-haiku@20241022", "name": "Claude Haiku 3.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "status": "deprecated", "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "claude-sonnet-4@20250514": { "id": "claude-sonnet-4@20250514", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "status": "deprecated", "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-opus-4-7@default": { "id": "claude-opus-4-7@default", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "claude-sonnet-4-5@20250929": { "id": "claude-sonnet-4-5@20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-sonnet-5@default": { "id": "claude-sonnet-5@default", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "claude-opus-4-6@default": { "id": "claude-opus-4-6@default", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "claude-opus-4-8@default": { "id": "claude-opus-4-8@default", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "claude-sonnet-4-6@default": { "id": "claude-sonnet-4-6@default", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } } } }, "v0": { "id": "v0", "env": ["V0_API_KEY"], "npm": "@ai-sdk/vercel", "name": "v0", "doc": "https://sdk.vercel.ai/providers/ai-sdk-providers/vercel", "models": { "v0-1.0-md": { "id": "v0-1.0-md", "name": "v0-1.0-md", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "v0", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 3, "output": 15 } }, "v0-1.5-lg": { "id": "v0-1.5-lg", "name": "v0-1.5-lg", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "v0", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-09", "last_updated": "2025-06-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 512000, "output": 32000 }, "cost": { "input": 15, "output": 75 } }, "v0-1.5-md": { "id": "v0-1.5-md", "name": "v0-1.5-md", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "v0", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-09", "last_updated": "2025-06-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 3, "output": 15 } } } }, "azure": { "id": "azure", "env": ["AZURE_RESOURCE_NAME", "AZURE_API_KEY"], "npm": "@ai-sdk/azure", "name": "Azure", "doc": "https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models", "models": { "codex-mini": { "id": "codex-mini", "name": "Codex Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-04", "release_date": "2025-05-16", "last_updated": "2025-05-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.5, "output": 6, "cache_read": 0.375 } }, "phi-3.5-moe-instruct": { "id": "phi-3.5-moe-instruct", "name": "Phi-3.5-MoE-instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-08-20", "last_updated": "2024-08-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.16, "output": 0.64 } }, "gpt-3.5-turbo-instruct": { "id": "gpt-3.5-turbo-instruct", "name": "GPT-3.5 Turbo Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-09-21", "last_updated": "2023-09-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 1.5, "output": 2 } }, "deepseek-r1-0528": { "id": "deepseek-r1-0528", "name": "DeepSeek-R1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 1.35, "output": 5.4 } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek-V4-Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models", "shape": "completions" }, "cost": { "input": 0.19, "output": 0.51 } }, "gpt-5.2-chat": { "id": "gpt-5.2-chat", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "o3": { "id": "o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "deepseek-v3-0324": { "id": "deepseek-v3-0324", "name": "DeepSeek-V3-0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 1.14, "output": 4.56 } }, "phi-3-small-128k-instruct": { "id": "phi-3-small-128k-instruct", "name": "Phi-3-small-instruct (128k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.15, "output": 0.6 } }, "meta-llama-3-8b-instruct": { "id": "meta-llama-3-8b-instruct", "name": "Meta-Llama-3-8B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0.3, "output": 0.61 } }, "mistral-small-2503": { "id": "mistral-small-2503", "name": "Mistral Small 3.1", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2025-03-01", "last_updated": "2025-03-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.3 } }, "text-embedding-3-large": { "id": "text-embedding-3-large", "name": "text-embedding-3-large", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 3072 }, "cost": { "input": 0.13, "output": 0 } }, "o1-mini": { "id": "o1-mini", "name": "o1-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-09-12", "last_updated": "2024-09-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 65536 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "phi-3.5-mini-instruct": { "id": "phi-3.5-mini-instruct", "name": "Phi-3.5-mini-instruct", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-08-20", "last_updated": "2024-08-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.13, "output": 0.52 } }, "mistral-nemo": { "id": "mistral-nemo", "name": "Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.15 } }, "cohere-embed-v-4-0": { "id": "cohere-embed-v-4-0", "name": "Embed v4", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "cohere-embed", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 1536 }, "cost": { "input": 0.12, "output": 0 } }, "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-08-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "gpt-5-chat": { "id": "gpt-5-chat", "name": "GPT-5 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2024-10-24", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "cohere-command-a": { "id": "cohere-command-a", "name": "Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8000 }, "cost": { "input": 2.5, "output": 10 } }, "llama-3.2-11b-vision-instruct": { "id": "llama-3.2-11b-vision-instruct", "name": "Llama-3.2-11B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.37, "output": 0.37 } }, "gpt-5-pro": { "id": "gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "cohere-command-r-08-2024": { "id": "cohere-command-r-08-2024", "name": "Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.15, "output": 0.6 } }, "gpt-4o": { "id": "gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "gpt-4": { "id": "gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-03-14", "last_updated": "2023-03-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 60, "output": 120 } }, "o4-mini": { "id": "o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "phi-3-medium-128k-instruct": { "id": "phi-3-medium-128k-instruct", "name": "Phi-3-medium-instruct (128k)", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.17, "output": 0.68 } }, "gpt-4-32k": { "id": "gpt-4-32k", "name": "GPT-4 32K", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-03-14", "last_updated": "2023-03-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 60, "output": 120 } }, "meta-llama-3.1-405b-instruct": { "id": "meta-llama-3.1-405b-instruct", "name": "Meta-Llama-3.1-405B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 5.33, "output": 16 } }, "cohere-command-r-plus-08-2024": { "id": "cohere-command-r-plus-08-2024", "name": "Command R+", "description": "Cohere's RAG workhorse for long-context enterprise search and tool use", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 2.5, "output": 10 } }, "phi-4-mini": { "id": "phi-4-mini", "name": "Phi-4-mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.075, "output": 0.3 } }, "gpt-3.5-turbo-1106": { "id": "gpt-3.5-turbo-1106", "name": "GPT-3.5 Turbo 1106", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-11-06", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 1, "output": 2 } }, "llama-4-scout-17b-16e-instruct": { "id": "llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout 17B 16E Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.2, "output": 0.78 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.71, "output": 0.71 } }, "grok-4-20-non-reasoning": { "id": "grok-4-20-non-reasoning", "name": "Grok 4.20 (Non-Reasoning)", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2026-04-08", "last_updated": "2026-04-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 8192 }, "status": "beta", "cost": { "input": 2, "output": 6 } }, "gpt-5.1-chat": { "id": "gpt-5.1-chat", "name": "GPT-5.1 Chat", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "image", "audio"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek-V4-Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models", "shape": "completions" }, "cost": { "input": 1.74, "output": 3.48 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "status": "beta", "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "deepseek-r1": { "id": "deepseek-r1", "name": "DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 1.35, "output": 5.4 } }, "grok-4-1-fast-non-reasoning": { "id": "grok-4-1-fast-non-reasoning", "name": "Grok 4.1 Fast (Non-Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-27", "last_updated": "2025-06-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "status": "beta", "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 Nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "phi-3-small-8k-instruct": { "id": "phi-3-small-8k-instruct", "name": "Phi-3-small-instruct (8k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0.15, "output": 0.6 } }, "gpt-5.3-chat": { "id": "gpt-5.3-chat", "name": "GPT-5.3 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-3.5-turbo-0125": { "id": "gpt-3.5-turbo-0125", "name": "GPT-3.5 Turbo 0125", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.5, "output": 1.5 } }, "cohere-embed-v3-multilingual": { "id": "cohere-embed-v3-multilingual", "name": "Embed v3 Multilingual", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "cohere-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-11-07", "last_updated": "2023-11-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 1024 }, "cost": { "input": 0.1, "output": 0 } }, "gpt-3.5-turbo-0613": { "id": "gpt-3.5-turbo-0613", "name": "GPT-3.5 Turbo 0613", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-06-13", "last_updated": "2023-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 3, "output": 4 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "image", "audio"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.1-codex-max": { "id": "gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "phi-4-reasoning": { "id": "phi-4-reasoning", "name": "Phi-4-reasoning", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.125, "output": 0.5 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-12-31", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "mistral-medium-2505": { "id": "mistral-medium-2505", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.4, "output": 2 } }, "meta-llama-3.1-70b-instruct": { "id": "meta-llama-3.1-70b-instruct", "name": "Meta-Llama-3.1-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 2.68, "output": 3.54 } }, "text-embedding-ada-002": { "id": "text-embedding-ada-002", "name": "text-embedding-ada-002", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2022-12-15", "last_updated": "2022-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 }, "cost": { "input": 0.1, "output": 0 } }, "o3-mini": { "id": "o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.125 } }, "gpt-4-turbo-vision": { "id": "gpt-4-turbo-vision", "name": "GPT-4 Turbo Vision", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "mistral-large-2411": { "id": "mistral-large-2411", "name": "Mistral Large 24.11", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 2, "output": 6 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "text-embedding-3-small": { "id": "text-embedding-3-small", "name": "text-embedding-3-small", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 1536 }, "cost": { "input": 0.02, "output": 0 } }, "deepseek-v3.2-speciale": { "id": "deepseek-v3.2-speciale", "name": "DeepSeek-V3.2-Speciale", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.58, "output": 1.68 } }, "gpt-3.5-turbo-0301": { "id": "gpt-3.5-turbo-0301", "name": "GPT-3.5 Turbo 0301", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-03-01", "last_updated": "2023-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 1.5, "output": 2 } }, "meta-llama-3.1-8b-instruct": { "id": "meta-llama-3.1-8b-instruct", "name": "Meta-Llama-3.1-8B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.61 } }, "claude-opus-4-1": { "id": "claude-opus-4-1", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "GPT-5.1 Codex Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-06", "last_updated": "2026-02-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models", "shape": "completions" }, "cost": { "input": 0.6, "output": 3 } }, "phi-4": { "id": "phi-4", "name": "Phi-4", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.125, "output": 0.5 } }, "grok-4-fast-reasoning": { "id": "grok-4-fast-reasoning", "name": "Grok 4 Fast (Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "gpt-image-1.5": { "id": "gpt-image-1.5", "name": "GPT-Image-1.5", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 32, "cache_read": 1.25 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "o1": { "id": "o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "deepseek-v3.1": { "id": "deepseek-v3.1", "name": "DeepSeek-V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.56, "output": 1.68 } }, "llama-3.2-90b-vision-instruct": { "id": "llama-3.2-90b-vision-instruct", "name": "Llama-3.2-90B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 2.04, "output": 2.04 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-31", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "ministral-3b": { "id": "ministral-3b", "name": "Ministral 3B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.04, "output": 0.04 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 Mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models", "shape": "completions" }, "cost": { "input": 0.95, "output": 4 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "grok-4-1-fast-reasoning": { "id": "grok-4-1-fast-reasoning", "name": "Grok 4.1 Fast (Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-27", "last_updated": "2025-06-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "status": "beta", "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.03 } }, "gpt-image-1": { "id": "gpt-image-1", "name": "GPT-Image-1", "description": "OpenAI image model for production generation, edits, and brand-safe visual workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-04-24", "last_updated": "2025-04-24", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 40, "cache_read": 1.25 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "gpt-4-turbo": { "id": "gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.01 } }, "phi-3-medium-4k-instruct": { "id": "phi-3-medium-4k-instruct", "name": "Phi-3-medium-instruct (4k)", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 1024 }, "cost": { "input": 0.17, "output": 0.68 } }, "gpt-5.4-pro": { "id": "gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "cohere-embed-v3-english": { "id": "cohere-embed-v3-english", "name": "Embed v3 English", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "cohere-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-11-07", "last_updated": "2023-11-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 1024 }, "cost": { "input": 0.1, "output": 0 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "codestral-2501": { "id": "codestral-2501", "name": "Codestral 25.01", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "codestral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.3, "output": 0.9 } }, "phi-4-reasoning-plus": { "id": "phi-4-reasoning-plus", "name": "Phi-4-reasoning-plus", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.125, "output": 0.5 } }, "phi-4-multimodal": { "id": "phi-4-multimodal", "name": "Phi-4-multimodal", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "phi", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.08, "output": 0.32, "input_audio": 4 } }, "gpt-4o-mini": { "id": "gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "model-router": { "id": "model-router", "name": "Model Router", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "model-router", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2025-05-19", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.14, "output": 0 } }, "grok-4-20-reasoning": { "id": "grok-4-20-reasoning", "name": "Grok 4.20 (Reasoning)", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2026-04-08", "last_updated": "2026-04-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 8192 }, "status": "beta", "cost": { "input": 2, "output": 6 } }, "gpt-5-codex": { "id": "gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.58, "output": 1.68 } }, "llama-4-maverick-17b-128e-instruct-fp8": { "id": "llama-4-maverick-17b-128e-instruct-fp8", "name": "Llama 4 Maverick 17B 128E Instruct FP8", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.25, "output": 1 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "phi-4-mini-reasoning": { "id": "phi-4-mini-reasoning", "name": "Phi-4-mini-reasoning", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.075, "output": 0.3 } }, "phi-3-mini-128k-instruct": { "id": "phi-3-mini-128k-instruct", "name": "Phi-3-mini-instruct (128k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.13, "output": 0.52 } }, "gpt-image-2": { "id": "gpt-image-2", "name": "GPT-Image-2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5, "output": 30, "cache_read": 1.25 } }, "phi-3-mini-4k-instruct": { "id": "phi-3-mini-4k-instruct", "name": "Phi-3-mini-instruct (4k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 1024 }, "cost": { "input": 0.13, "output": 0.52 } }, "meta-llama-3-70b-instruct": { "id": "meta-llama-3-70b-instruct", "name": "Meta-Llama-3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 2.68, "output": 3.54 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "image", "audio"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "cerebras": { "id": "cerebras", "env": ["CEREBRAS_API_KEY"], "npm": "@ai-sdk/cerebras", "name": "Cerebras", "doc": "https://inference-docs.cerebras.ai/models/overview", "models": { "gemma-4-31b": { "id": "gemma-4-31b", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-07-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 40960 }, "status": "beta", "cost": { "input": 0.99, "output": 1.49 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2026-06-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 40960 }, "cost": { "input": 0.35, "output": 0.75 } }, "zai-glm-4.7": { "id": "zai-glm-4.7", "name": "Z.AI GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-07", "last_updated": "2026-06-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 40960 }, "status": "beta", "cost": { "input": 2.25, "output": 2.75, "cache_read": 0, "cache_write": 0 } } } }, "zai-coding-plan": { "id": "zai-coding-plan", "env": ["ZHIPU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.z.ai/api/coding/paas/v4", "name": "Z.AI Coding Plan", "doc": "https://docs.z.ai/devpack/overview", "models": { "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "GLM-4.5-Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5-turbo": { "id": "glm-5-turbo", "name": "GLM-5-Turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "nvidia": { "id": "nvidia", "env": ["NVIDIA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://integrate.api.nvidia.com/v1", "name": "Nvidia", "doc": "https://docs.api.nvidia.com/nim/", "models": { "baai/bge-m3": { "id": "baai/bge-m3", "name": "BGE M3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-01-30", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 1024 }, "cost": { "input": 0, "output": 0 } }, "moonshotai/kimi-k2-instruct-0905": { "id": "moonshotai/kimi-k2-instruct-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0, "output": 0 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0, "output": 0 } }, "minimaxai/minimax-m3": { "id": "minimaxai/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "minimaxai/minimax-m2.7": { "id": "minimaxai/minimax-m2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "stepfun-ai/step-3.7-flash": { "id": "stepfun-ai/step-3.7-flash", "name": "Step 3.7 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "stepfun-ai/step-3.5-flash": { "id": "stepfun-ai/step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-02", "last_updated": "2026-02-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "google/gemma-3n-e4b-it": { "id": "google/gemma-3n-e4b-it", "name": "Gemma 3n E4b It", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-06-03", "last_updated": "2025-06-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "google/gemma-3n-e2b-it": { "id": "google/gemma-3n-e2b-it", "name": "Gemma 3n E2b It", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-06-12", "last_updated": "2025-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "google/google-paligemma": { "id": "google/google-paligemma", "name": "paligemma", "description": "Gemini multimodal model for text, image, audio, video, and document tasks", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-05-14", "last_updated": "2024-08-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma-4-31B-IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "google/gemma-2-2b-it": { "id": "google/gemma-2-2b-it", "name": "Gemma 2 2b It", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-07-16", "last_updated": "2024-07-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-4-mini-instruct": { "id": "microsoft/phi-4-mini-instruct", "name": "Phi-4-Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-01", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "microsoft/phi-4-multimodal-instruct": { "id": "microsoft/phi-4-multimodal-instruct", "name": "Phi 4 Multimodal", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "z-ai/glm-5.2": { "id": "z-ai/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT-OSS-120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-04", "last_updated": "2025-08-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "openai/whisper-large-v3": { "id": "openai/whisper-large-v3", "name": "Whisper Large v3", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2023-09", "release_date": "2023-09-01", "last_updated": "2025-09-05", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "bytedance/seed-oss-36b-instruct": { "id": "bytedance/seed-oss-36b-instruct", "name": "ByteDance-Seed/Seed-OSS-36B-Instruct", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0, "output": 0 } }, "mistralai/mistral-7b-instruct-v03": { "id": "mistralai/mistral-7b-instruct-v03", "name": "Mistral-7B-Instruct-v0.3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-01", "last_updated": "2025-04-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "mistralai/magistral-small-2506": { "id": "mistralai/magistral-small-2506", "name": "Magistral Small 2506", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "mistralai/mixtral-8x7b-instruct": { "id": "mistralai/mixtral-8x7b-instruct", "name": "Mistral: Mixtral 8x7B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-12-10", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "mistralai/mistral-medium-3-instruct": { "id": "mistralai/mistral-medium-3-instruct", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "mistralai/mistral-small-4-119b-2603": { "id": "mistralai/mistral-small-4-119b-2603", "name": "mistral-small-4-119b-2603", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mistralai/mistral-nemotron": { "id": "mistralai/mistral-nemotron", "name": "mistral-nemotron", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-11", "last_updated": "2025-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mistralai/mistral-large-3-675b-instruct-2512": { "id": "mistralai/mistral-large-3-675b-instruct-2512", "name": "Mistral Large 3 675B Instruct 2512", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0 } }, "mistralai/mixtral-8x22b-instruct": { "id": "mistralai/mixtral-8x22b-instruct", "name": "Mistral: Mixtral 8x22B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-04-17", "last_updated": "2024-04-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 13108 }, "cost": { "input": 0, "output": 0 } }, "nvidia/cosmos-transfer1-7b": { "id": "nvidia/cosmos-transfer1-7b", "name": "cosmos-transfer1-7b", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-06-13", "last_updated": "2025-06-30", "modalities": { "input": ["text", "image", "video"], "output": ["video"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/cosmos-transfer2_5-2b": { "id": "nvidia/cosmos-transfer2_5-2b", "name": "cosmos-transfer2.5-2b", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "video"], "output": ["video"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/llama-nemotron-embed-vl-1b-v2": { "id": "nvidia/llama-nemotron-embed-vl-1b-v2", "name": "llama-nemotron-embed-vl-1b-v2", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "nemotron", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-02-10", "last_updated": "2026-02-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": { "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "name": "Nemotron 3 Nano Omni", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": -1, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "nvidia/magpie-tts-zeroshot": { "id": "nvidia/magpie-tts-zeroshot", "name": "magpie-tts-zeroshot", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-22", "last_updated": "2025-06-12", "modalities": { "input": ["text", "audio"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nvidia-nemotron-nano-9b-v2": { "id": "nvidia/nvidia-nemotron-nano-9b-v2", "name": "nvidia-nemotron-nano-9b-v2", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2025-08-18", "last_updated": "2025-08-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "nvidia/synthetic-video-detector": { "id": "nvidia/synthetic-video-detector", "name": "synthetic-video-detector", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-content-safety-reasoning-4b": { "id": "nvidia/nemotron-content-safety-reasoning-4b", "name": "nemotron-content-safety-reasoning-4b", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nv-embed-v1": { "id": "nvidia/nv-embed-v1", "name": "nv-embed-v1", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-06-07", "last_updated": "2025-07-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "nvidia/usdcode": { "id": "nvidia/usdcode", "name": "usdcode", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-01", "last_updated": "2026-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/riva-translate-4b-instruct-v1_1": { "id": "nvidia/riva-translate-4b-instruct-v1_1", "name": "riva-translate-4b-instruct-v1_1", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-12", "last_updated": "2025-12-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/sparsedrive": { "id": "nvidia/sparsedrive", "name": "sparsedrive", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-18", "last_updated": "2025-07-20", "modalities": { "input": ["video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "nvidia/rerank-qa-mistral-4b": { "id": "nvidia/rerank-qa-mistral-4b", "name": "rerank-qa-mistral-4b", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-03-17", "last_updated": "2025-01-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/streampetr": { "id": "nvidia/streampetr", "name": "streampetr", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "nvidia/active-speaker-detection": { "id": "nvidia/active-speaker-detection", "name": "Active Speaker Detection", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/llama-3_1-nemotron-safety-guard-8b-v3": { "id": "nvidia/llama-3_1-nemotron-safety-guard-8b-v3", "name": "llama-3.1-nemotron-safety-guard-8b-v3", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/llama-3_2-nemoretriever-300m-embed-v1": { "id": "nvidia/llama-3_2-nemoretriever-300m-embed-v1", "name": "llama-3_2-nemoretriever-300m-embed-v1", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-07-24", "last_updated": "2025-07-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-voicechat": { "id": "nvidia/nemotron-voicechat", "name": "nemotron-voicechat", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nv-embedcode-7b-v1": { "id": "nvidia/nv-embedcode-7b-v1", "name": "nv-embedcode-7b-v1", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-03-17", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-ultra-550b-a55b": { "id": "nvidia/nemotron-3-ultra-550b-a55b", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 2.5, "cache_read": 0.15 } }, "nvidia/nemotron-3-nano-30b-a3b": { "id": "nvidia/nemotron-3-nano-30b-a3b", "name": "nemotron-3-nano-30b-a3b", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-12", "last_updated": "2024-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "nvidia/cosmos-predict1-5b": { "id": "nvidia/cosmos-predict1-5b", "name": "cosmos-predict1-5b", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-03-18", "last_updated": "2025-03-18", "modalities": { "input": ["text", "image", "video"], "output": ["video"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/bevformer": { "id": "nvidia/bevformer", "name": "bevformer", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-18", "last_updated": "2025-07-20", "modalities": { "input": ["video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "nvidia/studiovoice": { "id": "nvidia/studiovoice", "name": "studiovoice", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-03", "last_updated": "2025-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "nvidia/gliner-pii": { "id": "nvidia/gliner-pii", "name": "gliner-pii", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-mini-4b-instruct": { "id": "nvidia/nemotron-mini-4b-instruct", "name": "nemotron-mini-4b-instruct", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-08-21", "last_updated": "2024-08-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "nvidia/llama-nemotron-rerank-vl-1b-v2": { "id": "nvidia/llama-nemotron-rerank-vl-1b-v2", "name": "llama-nemotron-rerank-vl-1b-v2", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "nemotron", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nemotron 3 Super", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2, "output": 0.8 } }, "nvidia/usdvalidate": { "id": "nvidia/usdvalidate", "name": "usdvalidate", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-07-24", "last_updated": "2025-01-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-content-safety": { "id": "nvidia/nemotron-3-content-safety", "name": "nemotron-3-content-safety", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "abacusai/dracarys-llama-3_1-70b-instruct": { "id": "abacusai/dracarys-llama-3_1-70b-instruct", "name": "dracarys-llama-3.1-70b-instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-09-11", "last_updated": "2025-05-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "deepseek-ai/deepseek-v4-flash": { "id": "deepseek-ai/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 393216 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "deepseek-ai/deepseek-v4-pro": { "id": "deepseek-ai/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 393216 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "qwen/qwen3-next-80b-a3b-instruct": { "id": "qwen/qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next-80B-A3B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-01", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen-image-edit": { "id": "qwen/qwen-image-edit", "name": "Qwen Image Edit", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen-image": { "id": "qwen/qwen-image", "name": "Qwen Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3-coder-480b-a35b-instruct": { "id": "qwen/qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 66536 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen2.5-coder-32b-instruct": { "id": "qwen/qwen2.5-coder-32b-instruct", "name": "Qwen2.5 Coder 32b Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-11-06", "last_updated": "2024-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3.5-397b-a17b": { "id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5-397B-A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3.5-122b-a10b": { "id": "qwen/qwen3.5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "sarvamai/sarvam-m": { "id": "sarvamai/sarvam-m", "name": "sarvam-m", "description": "Efficient Indian-language reasoning model for chat, coding, and multilingual work", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.1-8b-instruct": { "id": "meta/llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.1-70b-instruct": { "id": "meta/llama-3.1-70b-instruct", "name": "Llama 3.1 70b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-07-16", "last_updated": "2024-07-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.2-1b-instruct": { "id": "meta/llama-3.2-1b-instruct", "name": "Llama 3.2 1b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.2-11b-vision-instruct": { "id": "meta/llama-3.2-11b-vision-instruct", "name": "Llama 3.2 11b Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.3-70b-instruct": { "id": "meta/llama-3.3-70b-instruct", "name": "Llama 3.3 70b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-11-26", "last_updated": "2024-11-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-guard-4-12b": { "id": "meta/llama-guard-4-12b", "name": "Llama Guard 4 12B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-05", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "meta/esmfold": { "id": "meta/esmfold", "name": "esmfold", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-03-15", "last_updated": "2025-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/esm2-650m": { "id": "meta/esm2-650m", "name": "esm2-650m", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-29", "last_updated": "2025-03-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.2-90b-vision-instruct": { "id": "meta/llama-3.2-90b-vision-instruct", "name": "Llama-3.2-90B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-4-maverick-17b-128e-instruct": { "id": "meta/llama-4-maverick-17b-128e-instruct", "name": "Llama 4 Maverick 17b 128e Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-02", "release_date": "2025-04-01", "last_updated": "2025-04-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "meta/llama-3.2-3b-instruct": { "id": "meta/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "upstage/solar-10_7b-instruct": { "id": "upstage/solar-10_7b-instruct", "name": "solar-10.7b-instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-06-05", "last_updated": "2025-04-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "black-forest-labs/flux_1-schnell": { "id": "black-forest-labs/flux_1-schnell", "name": "FLUX.1-schnell", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "knowledge": "2024-07", "release_date": "2024-08-01", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": true, "limit": { "context": 77, "input": 77, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "black-forest-labs/flux_1-kontext-dev": { "id": "black-forest-labs/flux_1-kontext-dev", "name": "FLUX.1-Kontext-dev", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": true, "limit": { "context": 40960, "output": 40960 }, "cost": { "input": 0, "output": 0 } }, "black-forest-labs/flux.1-dev": { "id": "black-forest-labs/flux.1-dev", "name": "FLUX.1-dev", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-08", "release_date": "2024-08-01", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 4096, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "black-forest-labs/flux_2-klein-4b": { "id": "black-forest-labs/flux_2-klein-4b", "name": "FLUX.2 Klein 4B", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-06", "release_date": "2026-01-14", "last_updated": "2026-01-31", "modalities": { "input": ["image", "text"], "output": ["image"] }, "open_weights": true, "limit": { "context": 40960, "output": 40960 }, "cost": { "input": 0, "output": 0 } } } }, "evroc": { "id": "evroc", "env": ["EVROC_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://models.think.evroc.com/v1", "name": "evroc", "doc": "https://docs.evroc.com/products/think/overview.html", "models": { "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.4375, "output": 5.75 } }, "google/gemma-4-26B-A4B-it": { "id": "google/gemma-4-26B-A4B-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.144, "output": 0.575 } }, "Qwen/Qwen3-Embedding-8B": { "id": "Qwen/Qwen3-Embedding-8B", "name": "Qwen3 Embedding 8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 4096 }, "cost": { "input": 0.115, "output": 0.115 } }, "Qwen/Qwen3-Reranker-4B": { "id": "Qwen/Qwen3-Reranker-4B", "name": "Qwen3 Reranker 4B", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.0575, "output": 0 } }, "Qwen/Qwen3.6-35B-A3B-FP8": { "id": "Qwen/Qwen3.6-35B-A3B-FP8", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.345, "output": 1.38 } }, "Qwen/Qwen3-VL-30B-A3B-Instruct": { "id": "Qwen/Qwen3-VL-30B-A3B-Instruct", "name": "Qwen3 VL 30B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 100000, "output": 100000 }, "cost": { "input": 0.23, "output": 0.92 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.23, "output": 0.92 } }, "openai/whisper-large-v3-turbo": { "id": "openai/whisper-large-v3-turbo", "name": "Whisper Large v3 Turbo", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 448, "output": 448 }, "cost": { "input": 0.0023, "output": 0.0023, "output_audio": 2.3 } }, "openai/whisper-large-v3": { "id": "openai/whisper-large-v3", "name": "Whisper 3 Large", "description": "Open Whisper checkpoint for robust multilingual transcription and captioning", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 448, "output": 4096 }, "cost": { "input": 0.0023, "output": 0.0023, "output_audio": 2.3 } }, "mistralai/Mistral-Medium-3.5-128B": { "id": "mistralai/Mistral-Medium-3.5-128B", "name": "Mistral Medium 3.5", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.725, "output": 6.9 } }, "mistralai/Voxtral-Small-24B-2507": { "id": "mistralai/Voxtral-Small-24B-2507", "name": "Voxtral Small 24B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "voxtral", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2025-03-01", "last_updated": "2025-03-01", "modalities": { "input": ["audio", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.0023, "output": 0.0023, "output_audio": 2.3 } }, "nvidia/Llama-3.3-70B-Instruct-FP8": { "id": "nvidia/Llama-3.3-70B-Instruct-FP8", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 1.15, "output": 1.15 } }, "evroc/roc": { "id": "evroc/roc", "name": "roc", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-06-06", "last_updated": "2026-06-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 2.875, "output": 11.516 } }, "KBLab/kb-whisper-large": { "id": "KBLab/kb-whisper-large", "name": "KB Whisper", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 448, "output": 448 }, "cost": { "input": 0.0023, "output": 0.0023, "output_audio": 2.3 } }, "intfloat/multilingual-e5-large-instruct": { "id": "intfloat/multilingual-e5-large-instruct", "name": "E5 Multi-Lingual Large Embeddings 0.6B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 512 }, "cost": { "input": 0.114, "output": 0.114 } } } }, "xiaomi": { "id": "xiaomi", "env": ["XIAOMI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.xiaomimimo.com/v1", "name": "Xiaomi", "doc": "https://platform.xiaomimimo.com/#/docs", "models": { "mimo-v2.5-pro-ultraspeed": { "id": "mimo-v2.5-pro-ultraspeed", "name": "MiMo-V2.5-Pro-UltraSpeed", "description": "MiMo pro model for strong multimodal reasoning and agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-06-08", "last_updated": "2026-06-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "status": "beta", "cost": { "input": 1.305, "output": 2.61, "cache_read": 0.0108 } }, "mimo-v2.5": { "id": "mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-06-24", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "mimo-v2-omni": { "id": "mimo-v2-omni", "name": "MiMo-V2-Omni", "description": "Legacy model retained for compatibility with older integrations", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-06-24", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "status": "deprecated", "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "mimo-v2-flash": { "id": "mimo-v2-flash", "name": "MiMo-V2-Flash", "description": "Legacy model retained for compatibility with older integrations", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12-01", "release_date": "2025-12-16", "last_updated": "2026-06-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "status": "deprecated", "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "mimo-v2-pro": { "id": "mimo-v2-pro", "name": "MiMo-V2-Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-06-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 131072 }, "status": "deprecated", "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036 } }, "mimo-v2.5-pro": { "id": "mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-06-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036 } } } }, "inception": { "id": "inception", "env": ["INCEPTION_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.inceptionlabs.ai/v1/", "name": "Inception", "doc": "https://platform.inceptionlabs.ai/docs", "models": { "mercury-edit-2": { "id": "mercury-edit-2", "name": "Mercury Edit 2", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-03-30", "last_updated": "2026-03-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.25, "output": 0.75, "cache_read": 0.025 } }, "mercury-2": { "id": "mercury-2", "name": "Mercury 2", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "mercury", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 50000 }, "cost": { "input": 0.25, "output": 0.75, "cache_read": 0.025 } } } }, "anthropic": { "id": "anthropic", "env": ["ANTHROPIC_API_KEY"], "npm": "@ai-sdk/anthropic", "name": "Anthropic", "doc": "https://docs.anthropic.com/en/docs/about-claude/models", "models": { "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "claude-opus-4-1-20250805": { "id": "claude-opus-4-1-20250805", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-14", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-29", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "claude-opus-4-5-20251101": { "id": "claude-opus-4-5-20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-opus-4-1": { "id": "claude-opus-4-1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-07", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-04", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } } } }, "tencent-coding-plan": { "id": "tencent-coding-plan", "env": ["TENCENT_CODING_PLAN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.lkeap.cloud.tencent.com/coding/v3", "name": "Tencent Coding Plan (China)", "doc": "https://cloud.tencent.com/document/product/1772/128947", "models": { "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "hunyuan-turbos": { "id": "hunyuan-turbos", "name": "Hunyuan-TurboS", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-08", "last_updated": "2026-03-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "hunyuan-t1": { "id": "hunyuan-t1", "name": "Hunyuan-T1", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-03-08", "last_updated": "2026-03-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "tc-code-latest": { "id": "tc-code-latest", "name": "Auto", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "auto", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-08", "last_updated": "2026-03-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "hunyuan-2.0-instruct": { "id": "hunyuan-2.0-instruct", "name": "Tencent HY 2.0 Instruct", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-08", "last_updated": "2026-03-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "hunyuan-2.0-thinking": { "id": "hunyuan-2.0-thinking", "name": "Tencent HY 2.0 Think", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-03-08", "last_updated": "2026-03-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "freemodel": { "id": "freemodel", "env": ["FREEMODEL_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://cc.freemodel.dev/v1", "name": "FreeModel", "doc": "https://freemodel.dev", "models": { "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://api.freemodel.dev/v1" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175, "cache_write": 1.75 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://api.freemodel.dev/v1" }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 2.5 } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://api.freemodel.dev/v1" }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075, "cache_write": 0.75 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://api.freemodel.dev/v1" }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 5 } } } }, "sap-ai-core": { "id": "sap-ai-core", "env": ["AICORE_SERVICE_KEY"], "npm": "@jerome-benoit/sap-ai-provider-v2", "name": "SAP AI Core", "doc": "https://help.sap.com/docs/sap-ai-core", "models": { "anthropic--claude-4.8-opus": { "id": "anthropic--claude-4.8-opus", "name": "anthropic--claude-4.8-opus", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gemini-3.1-flash-lite": { "id": "gemini-3.1-flash-lite", "name": "gemini-3.1-flash-lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "anthropic--claude-4.6-sonnet": { "id": "anthropic--claude-4.6-sonnet", "name": "anthropic--claude-4.6-sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic--claude-3-sonnet": { "id": "anthropic--claude-3-sonnet", "name": "anthropic--claude-3-sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-03-04", "last_updated": "2024-03-04", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic--claude-4-sonnet": { "id": "anthropic--claude-4-sonnet", "name": "anthropic--claude-4-sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "gemini-2.5-pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-03-25", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-5": { "id": "gpt-5", "name": "gpt-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "gemini-2.5-flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-04-17", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "gemini-3.5-flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "anthropic--claude-4.5-haiku": { "id": "anthropic--claude-4.5-haiku", "name": "anthropic--claude-4.5-haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic--claude-3-haiku": { "id": "anthropic--claude-3-haiku", "name": "anthropic--claude-3-haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-03-13", "last_updated": "2024-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.03, "cache_write": 0.3 } }, "anthropic--claude-4-opus": { "id": "anthropic--claude-4-opus", "name": "anthropic--claude-4-opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic--claude-4.5-sonnet": { "id": "anthropic--claude-4.5-sonnet", "name": "anthropic--claude-4.5-sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic--claude-3.5-sonnet": { "id": "anthropic--claude-3.5-sonnet", "name": "anthropic--claude-3.5-sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic--claude-4.6-opus": { "id": "anthropic--claude-4.6-opus", "name": "anthropic--claude-4.6-opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "gemini-2.5-flash-lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01, "input_audio": 0.3 } }, "anthropic--claude-3.7-sonnet": { "id": "anthropic--claude-3.7-sonnet", "name": "anthropic--claude-3.7-sonnet", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2024-10-31", "release_date": "2025-02-24", "last_updated": "2025-02-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "sonar": { "id": "sonar", "name": "sonar", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 1, "output": 1 } }, "sonar-pro": { "id": "sonar-pro", "name": "sonar-pro", "description": "Advanced Sonar search model for deeper research and cited synthesis", "family": "sonar-pro", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 3, "output": 15 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "gpt-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "gpt-4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "anthropic--claude-3-opus": { "id": "anthropic--claude-3-opus", "name": "anthropic--claude-3-opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-02-29", "last_updated": "2024-02-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "gpt-5-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "gpt-4.1-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "sonar-deep-research": { "id": "sonar-deep-research", "name": "sonar-deep-research", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar-deep-research", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "temperature": false, "knowledge": "2025-01", "release_date": "2025-02-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 2, "output": 8, "reasoning": 3 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "gpt-5-nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "anthropic--claude-4.5-opus": { "id": "anthropic--claude-4.5-opus", "name": "anthropic--claude-4.5-opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic--claude-4.7-opus": { "id": "anthropic--claude-4.7-opus", "name": "anthropic--claude-4.7-opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "gpt-5.5", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } } } }, "opencode": { "id": "opencode", "env": ["OPENCODE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://opencode.ai/zen/v1", "name": "OpenCode Zen", "doc": "https://opencode.ai/docs/zen", "models": { "ring-2.6-1t-free": { "id": "ring-2.6-1t-free", "name": "Ring 2.6 1T Free", "description": "Legacy model retained for compatibility with older integrations", "family": "ring-1t-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-06", "release_date": "2026-05-08", "last_updated": "2026-05-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 66000 }, "status": "deprecated", "cost": { "input": 0, "output": 0 } }, "mimo-v2-pro-free": { "id": "mimo-v2-pro-free", "name": "MiMo V2 Pro Free", "description": "Legacy model retained for compatibility with older integrations", "family": "mimo-pro-free", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 64000 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Legacy model retained for compatibility with older integrations", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.1 } }, "mimo-v2.5-free": { "id": "mimo-v2.5-free", "name": "MiMo V2.5 Free", "description": "MiMo omni model for text, image, video, audio, and agents", "family": "mimo-v2.5-free", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "kimi-k2": { "id": "kimi-k2", "name": "Kimi K2", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2.5, "cache_read": 0.4 } }, "minimax-m2.1": { "id": "minimax-m2.1", "name": "MiniMax-M2.1", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.1 } }, "nemotron-3-ultra-free": { "id": "nemotron-3-ultra-free", "name": "Nemotron 3 Ultra Free", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2026-02", "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "glm-4.7-free": { "id": "glm-4.7-free", "name": "GLM-4.7 Free", "description": "Legacy model retained for compatibility with older integrations", "family": "glm-free", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "gemini-3-flash": { "id": "gemini-3-flash", "name": "Gemini 3 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "provider": { "npm": "@ai-sdk/google" }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05 } }, "deepseek-v4-flash-free": { "id": "deepseek-v4-flash-free", "name": "DeepSeek V4 Flash Free", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek-flash-free", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "claude-sonnet-4": { "id": "claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.07, "output": 8.5, "cache_read": 0.107 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "provider": { "npm": "@ai-sdk/google" }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "minimax-m3-free": { "id": "minimax-m3-free", "name": "MiniMax-M3 Free", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax-m3-free", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-31", "last_updated": "2026-05-31", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "gemini-3-pro": { "id": "gemini-3-pro", "name": "Gemini 3 Pro", "description": "Legacy model retained for compatibility with older integrations", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/google" }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "kimi-k2.5-free": { "id": "kimi-k2.5-free", "name": "Kimi K2.5 Free", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-free", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-10", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.74, "output": 3.84, "cache_read": 0.145 } }, "glm-4.6": { "id": "glm-4.6", "name": "GLM-4.6", "description": "Legacy model retained for compatibility with older integrations", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.1 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2.5, "cache_read": 0.4 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "qwen3-coder": { "id": "qwen3-coder", "name": "Qwen3 Coder", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "status": "deprecated", "cost": { "input": 0.45, "output": 1.8 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "hy3-preview-free": { "id": "hy3-preview-free", "name": "Hy3 preview Free", "description": "Legacy model retained for compatibility with older integrations", "family": "hy3-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "minimax-m3": { "id": "minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.07, "output": 8.5, "cache_read": 0.107 } }, "gpt-5.3-codex-spark": { "id": "gpt-5.3-codex-spark", "name": "GPT-5.3 Codex Spark", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex-spark", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.1-codex-max": { "id": "gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "ling-2.6-flash-free": { "id": "ling-2.6-flash-free", "name": "Ling 2.6 Flash Free", "description": "Legacy model retained for compatibility with older integrations", "family": "ling-flash-free", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262100, "output": 32800 }, "status": "deprecated", "cost": { "input": 0, "output": 0 } }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "qwen3.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.2, "output": 1.2, "cache_read": 0.02, "cache_write": 0.25 } }, "claude-3-5-haiku": { "id": "claude-3-5-haiku", "name": "Claude Haiku 3.5", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "minimax-m2.7": { "id": "minimax-m2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "grok-4.5": { "id": "grok-4.5", "name": "Grok 4.5", "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 500000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.5, "tiers": [ { "input": 4, "output": 12, "cache_read": 1, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 12, "cache_read": 1 } } }, "north-mini-code-free": { "id": "north-mini-code-free", "name": "North Mini Code Free", "description": "Cohere coding model for practical software engineering and agentic edits", "family": "north-free", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-09-23", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "grok-code": { "id": "grok-code", "name": "Grok Code Fast 1", "description": "Legacy model retained for compatibility with older integrations", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-20", "last_updated": "2025-08-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25, "tiers": [ { "input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5 } } }, "claude-opus-4-1": { "id": "claude-opus-4-1", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "GPT-5.1 Codex Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-10", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.08 } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 3.125, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 6.25, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5, "cache_write": 6.25 } } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "hy3-free": { "id": "hy3-free", "name": "Hy3 Free", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hy3-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-26", "last_updated": "2026-06-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 190000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "minimax-m2.5-free": { "id": "minimax-m2.5-free", "name": "MiniMax-M2.5 Free", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-10", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "minimax-m2.1-free": { "id": "minimax-m2.1-free", "name": "MiniMax-M2.1 Free", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "gpt-5.4-pro": { "id": "gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 30, "output": 180, "cache_read": 30 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "mimo-v2-flash-free": { "id": "mimo-v2-flash-free", "name": "MiMo V2 Flash Free", "description": "Legacy model retained for compatibility with older integrations", "family": "mimo-flash-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "gemini-3.1-pro": { "id": "gemini-3.1-pro", "name": "Gemini 3.1 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "provider": { "npm": "@ai-sdk/google" }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "trinity-large-preview-free": { "id": "trinity-large-preview-free", "name": "Trinity Large Preview", "description": "Legacy model retained for compatibility with older integrations", "family": "trinity", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-01-28", "last_updated": "2026-01-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "status": "deprecated", "cost": { "input": 0, "output": 0 } }, "big-pickle": { "id": "big-pickle", "name": "Big Pickle", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "big-pickle", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-10-17", "last_updated": "2025-10-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 160000, "output": 32000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "gpt-5.5-pro": { "id": "gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 30, "output": 180, "cache_read": 30 } }, "qwen3.6-plus-free": { "id": "qwen3.6-plus-free", "name": "Qwen3.6 Plus Free", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen-free", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "mimo-v2-omni-free": { "id": "mimo-v2-omni-free", "name": "MiMo V2 Omni Free", "description": "Legacy model retained for compatibility with older integrations", "family": "mimo-omni-free", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text", "image", "audio", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 64000 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5 } } }, "gpt-5-codex": { "id": "gpt-5-codex", "name": "GPT-5 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.07, "output": 8.5, "cache_read": 0.107 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "glm-5-free": { "id": "glm-5-free", "name": "GLM-5 Free", "description": "Legacy model retained for compatibility with older integrations", "family": "glm-free", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "provider": { "npm": "@ai-sdk/anthropic" }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625 } }, "nemotron-3-super-free": { "id": "nemotron-3-super-free", "name": "Nemotron 3 Super Free", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2026-02", "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128000 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "grok-build-0.1": { "id": "grok-build-0.1", "name": "Grok Build 0.1", "description": "Grok coding model for agentic engineering, edits, and codebase workflows", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-20", "last_updated": "2026-05-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 1.07, "output": 8.5, "cache_read": 0.107 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai" }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "inference": { "id": "inference", "env": ["INFERENCE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://inference.net/v1", "name": "Inference", "doc": "https://inference.net/models", "models": { "mistral/mistral-nemo-12b-instruct": { "id": "mistral/mistral-nemo-12b-instruct", "name": "Mistral Nemo 12B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0.038, "output": 0.1 } }, "google/gemma-3": { "id": "google/gemma-3", "name": "Google Gemma 3", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 125000, "output": 4096 }, "cost": { "input": 0.15, "output": 0.3 } }, "osmosis/osmosis-structure-0.6b": { "id": "osmosis/osmosis-structure-0.6b", "name": "Osmosis Structure 0.6B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "osmosis", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4000, "output": 2048 }, "cost": { "input": 0.1, "output": 0.5 } }, "qwen/qwen3-embedding-4b": { "id": "qwen/qwen3-embedding-4b", "name": "Qwen 3 Embedding 4B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 2048 }, "cost": { "input": 0.01, "output": 0 } }, "qwen/qwen-2.5-7b-vision-instruct": { "id": "qwen/qwen-2.5-7b-vision-instruct", "name": "Qwen 2.5 7B Vision Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 125000, "output": 4096 }, "cost": { "input": 0.2, "output": 0.2 } }, "meta/llama-3.1-8b-instruct": { "id": "meta/llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0.025, "output": 0.025 } }, "meta/llama-3.2-1b-instruct": { "id": "meta/llama-3.2-1b-instruct", "name": "Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0.01, "output": 0.01 } }, "meta/llama-3.2-11b-vision-instruct": { "id": "meta/llama-3.2-11b-vision-instruct", "name": "Llama 3.2 11B Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0.055, "output": 0.055 } }, "meta/llama-3.2-3b-instruct": { "id": "meta/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0.02, "output": 0.02 } } } }, "inceptron": { "id": "inceptron", "env": ["INCEPTRON_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.inceptron.io/v1", "name": "Inceptron", "doc": "https://docs.inceptron.io", "models": { "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.66, "output": 3.5, "cache_read": 0.2, "cache_write": 0 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.75, "output": 3.5, "cache_read": 0.2, "cache_write": 0 } }, "moonshotai/Kimi-K2.6-Fast": { "id": "moonshotai/Kimi-K2.6-Fast", "name": "Kimi K2.6 Fast", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "alpha", "cost": { "input": 1.32, "output": 7, "cache_read": 0.4, "cache_write": 0 } }, "zai-org/GLM-5.1-FP8": { "id": "zai-org/GLM-5.1-FP8", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.2, "output": 4.2, "cache_read": 0.26, "cache_write": 0 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 }, "cost": { "input": 0.15, "output": 0.9, "cache_read": 0.05, "cache_write": 0 } } } }, "llama": { "id": "llama", "env": ["LLAMA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.llama.com/compat/v1/", "name": "Llama", "doc": "https://llama.developer.meta.com/docs/models", "models": { "llama-4-scout-17b-16e-instruct-fp8": { "id": "llama-4-scout-17b-16e-instruct-fp8", "name": "Llama-4-Scout-17B-16E-Instruct-FP8", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "cerebras-llama-4-maverick-17b-128e-instruct": { "id": "cerebras-llama-4-maverick-17b-128e-instruct", "name": "Cerebras-Llama-4-Maverick-17B-128E-Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "groq-llama-4-maverick-17b-128e-instruct": { "id": "groq-llama-4-maverick-17b-128e-instruct", "name": "Groq-Llama-4-Maverick-17B-128E-Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "cerebras-llama-4-scout-17b-16e-instruct": { "id": "cerebras-llama-4-scout-17b-16e-instruct", "name": "Cerebras-Llama-4-Scout-17B-16E-Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "llama-3.3-8b-instruct": { "id": "llama-3.3-8b-instruct", "name": "Llama-3.3-8B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "llama-4-maverick-17b-128e-instruct-fp8": { "id": "llama-4-maverick-17b-128e-instruct-fp8", "name": "Llama-4-Maverick-17B-128E-Instruct-FP8", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0, "output": 0 } } } }, "llmtr": { "id": "llmtr", "env": ["LLMTR_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://llmtr.com/v1", "name": "LLMTR", "doc": "https://llmtr.com/docs", "models": { "sincap": { "id": "sincap", "name": "Sincap", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-05", "last_updated": "2026-05-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "magibu-11b-v8": { "id": "magibu-11b-v8", "name": "Magibu 11B v8", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-06-05", "last_updated": "2026-06-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "gemma-4": { "id": "gemma-4", "name": "Gemma 4", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 5, "output": 10 } }, "medgemma-4b": { "id": "medgemma-4b", "name": "MedGemma 4B", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-26", "last_updated": "2026-04-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 4096 }, "cost": { "input": 3, "output": 5 } }, "qwen3-6-35b": { "id": "qwen3-6-35b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 65536 }, "cost": { "input": 5, "output": 10 } }, "trendyol-7b": { "id": "trendyol-7b", "name": "Trendyol 7B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-06-06", "last_updated": "2026-06-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0, "output": 0 } } } }, "cohere": { "id": "cohere", "env": ["COHERE_API_KEY"], "npm": "@ai-sdk/cohere", "name": "Cohere", "doc": "https://docs.cohere.com/docs/models", "models": { "c4ai-aya-expanse-32b": { "id": "c4ai-aya-expanse-32b", "name": "Aya Expanse 32B", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-24", "last_updated": "2024-10-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 } }, "command-a-03-2025": { "id": "command-a-03-2025", "name": "Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8000 }, "cost": { "input": 2.5, "output": 10 } }, "c4ai-aya-vision-32b": { "id": "c4ai-aya-vision-32b", "name": "Aya Vision 32B", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-04", "last_updated": "2025-05-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4000 } }, "command-r7b-arabic-02-2025": { "id": "command-r7b-arabic-02-2025", "name": "Command R7B Arabic", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.0375, "output": 0.15 } }, "c4ai-aya-vision-8b": { "id": "c4ai-aya-vision-8b", "name": "Aya Vision 8B", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-04", "last_updated": "2025-05-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4000 } }, "command-r-08-2024": { "id": "command-r-08-2024", "name": "Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.15, "output": 0.6 } }, "command-r7b-12-2024": { "id": "command-r7b-12-2024", "name": "Command R7B", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-12-02", "last_updated": "2024-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.0375, "output": 0.15 } }, "command-a-vision-07-2025": { "id": "command-a-vision-07-2025", "name": "Command A Vision", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8000 }, "cost": { "input": 2.5, "output": 10 } }, "command-a-plus-05-2026": { "id": "command-a-plus-05-2026", "name": "Command A Plus", "description": "Cohere's stronger command model for multilingual agents and enterprise workflows", "family": "command-a", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04-01", "release_date": "2026-05-20", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 2.5, "output": 10 } }, "command-a-translate-08-2025": { "id": "command-a-translate-08-2025", "name": "Command A Translate", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "command-a", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "output": 8000 }, "cost": { "input": 2.5, "output": 10 } }, "command-r-plus-08-2024": { "id": "command-r-plus-08-2024", "name": "Command R+", "description": "Cohere's RAG workhorse for long-context enterprise search and tool use", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 2.5, "output": 10 } }, "command-a-reasoning-08-2025": { "id": "command-a-reasoning-08-2025", "name": "Command A Reasoning", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1 } ], "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 2.5, "output": 10 } }, "north-mini-code-1-0": { "id": "north-mini-code-1-0", "name": "North Mini Code", "description": "Cohere coding model for practical software engineering and agentic edits", "family": "north", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-09-23", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://api.cohere.ai/compatibility/v1" }, "cost": { "input": 0, "output": 0 } }, "c4ai-aya-expanse-8b": { "id": "c4ai-aya-expanse-8b", "name": "Aya Expanse 8B", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-24", "last_updated": "2024-10-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8000, "output": 4000 } } } }, "sarvam": { "id": "sarvam", "env": ["SARVAM_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.sarvam.ai/v1", "name": "Sarvam AI", "doc": "https://docs.sarvam.ai/api-reference-docs/getting-started/models", "models": { "sarvam-105b": { "id": "sarvam-105b", "name": "Sarvam-105B", "description": "Flagship Indian-language reasoning model for enterprise multilingual applications", "family": "sarvam", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": [null, "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-18", "last_updated": "2026-03-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 } }, "sarvam-30b": { "id": "sarvam-30b", "name": "Sarvam-30B", "description": "Efficient Indian-language reasoning model for chat, coding, and multilingual work", "family": "sarvam", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": [null, "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-18", "last_updated": "2026-03-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 } } } }, "stepfun": { "id": "stepfun", "env": ["STEPFUN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.stepfun.com/v1", "name": "StepFun", "doc": "https://platform.stepfun.com/docs/zh/overview/concept", "models": { "step-1-32k": { "id": "step-1-32k", "name": "Step 1 (32K)", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-01-01", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 2.05, "output": 9.59, "cache_read": 0.41 } }, "step-3.7-flash": { "id": "step-3.7-flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-06-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 1.15, "cache_read": 0.04 } }, "step-3.5-flash-2603": { "id": "step-3.5-flash-2603", "name": "Step 3.5 Flash 2603", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "stepaudio-2.5-tts": { "id": "stepaudio-2.5-tts", "name": "StepAudio 2.5 TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "step", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "stepaudio-2.5-asr": { "id": "stepaudio-2.5-asr", "name": "StepAudio 2.5 ASR", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "step", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-24", "last_updated": "2026-07-02", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "step-3.5-flash": { "id": "step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "step-tts-2": { "id": "step-tts-2", "name": "Step TTS 2", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "step", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-03-01", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "step-2-16k": { "id": "step-2-16k", "name": "Step 2 (16K)", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-01-01", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 5.21, "output": 16.44, "cache_read": 1.04 } } } }, "hpc-ai": { "id": "hpc-ai", "env": ["HPC_AI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.hpc-ai.com/inference/v1", "name": "HPC-AI", "doc": "https://www.hpc-ai.com/doc/docs/quickstart/", "models": { "moonshotai/kimi-k2.7-code": { "id": "moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5 } }, "zai-org/glm-5.1": { "id": "zai-org/glm-5.1", "name": "GLM 5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202000, "output": 202000 }, "cost": { "input": 0.615, "output": 2.46, "cache_read": 0.133 } }, "zai-org/glm-5.2": { "id": "zai-org/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 128000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1002000, "output": 128000 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.145 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196000, "output": 195000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } } } }, "minimax-cn": { "id": "minimax-cn", "env": ["MINIMAX_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://api.minimaxi.com/anthropic/v1", "name": "MiniMax (minimaxi.com)", "doc": "https://platform.minimaxi.com/docs/guides/quickstart", "models": { "MiniMax-M2.1": { "id": "MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMax-M2.5-highspeed": { "id": "MiniMax-M2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "MiniMax-M2.7-highspeed": { "id": "MiniMax-M2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "MiniMax-M2": { "id": "MiniMax-M2", "name": "MiniMax-M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "MiniMax-M3": { "id": "MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "tiers": [ { "input": 0.6, "output": 2.4, "cache_read": 0.12, "tier": { "type": "context", "size": 512000 } } ], "context_over_200k": { "input": 0.6, "output": 2.4, "cache_read": 0.12 } } }, "MiniMax-M2.7": { "id": "MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } } } }, "unorouter": { "id": "unorouter", "env": ["UNOROUTER_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.unorouter.com/v1", "name": "UnoRouter", "doc": "https://unorouter.com/models", "models": { "gpt-5.5:free": { "id": "gpt-5.5:free", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.0625, "output": 0.125 } }, "gpt-5.4:free": { "id": "gpt-5.4:free", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "glm-4.5-flash:free": { "id": "glm-4.5-flash:free", "name": "GLM-4.5-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0, "output": 0 } }, "minimax-m2.7:free": { "id": "minimax-m2.7:free", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "step-3.7-flash:free": { "id": "step-3.7-flash:free", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0, "output": 0 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1.2, "output": 6 } }, "qwen3.5-397b-a17b:free": { "id": "qwen3.5-397b-a17b:free", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "nemotron-3-ultra-550b-a55b:free": { "id": "nemotron-3-ultra-550b-a55b:free", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1857, "output": 1.1142 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.8999, "output": 1.7999 } }, "deepseek-v4-flash:free": { "id": "deepseek-v4-flash:free", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.44, "output": 7.2 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.6001, "output": 5.0288 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.425, "output": 2.125 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.05, "output": 8.4 } }, "minimax-m2.7": { "id": "minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.819, "output": 3.276 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 1.8, "output": 10.8 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.2675, "output": 5.3368 } }, "glm-5.2:free": { "id": "glm-5.2:free", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "gemma-4-31b-it:free": { "id": "gemma-4-31b-it:free", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "deepseek-v4-pro:free": { "id": "deepseek-v4-pro:free", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 0.1875, "output": 1.125 } } } }, "alibaba-coding-plan": { "id": "alibaba-coding-plan", "env": ["ALIBABA_CODING_PLAN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://coding-intl.dashscope.aliyuncs.com/v1", "name": "Alibaba Coding Plan", "doc": "https://www.alibabacloud.com/help/en/model-studio/coding-plan", "models": { "qwen3-coder-plus": { "id": "qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.5, "cache_write": 3.125 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1875, "output": 1.125, "cache_write": 0.234375 } }, "qwen3-max-2026-01-23": { "id": "qwen3-max-2026-01-23", "name": "Qwen3 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-23", "last_updated": "2026-01-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "input": 196601, "output": 24576 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3-coder-next": { "id": "qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "longcat": { "id": "longcat", "env": ["LONGCAT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.longcat.chat/openai", "name": "LongCat", "doc": "https://longcat.chat/platform/docs/", "models": { "LongCat-2.0": { "id": "LongCat-2.0", "name": "LongCat-2.0", "description": "Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window", "family": "longcat", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.75, "output": 2.95, "cache_read": 0.015 } } } }, "poe": { "id": "poe", "env": ["POE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.poe.com/v1", "name": "Poe", "doc": "https://creator.poe.com/docs/external-applications/openai-compatible-api", "models": { "trytako/tako": { "id": "trytako/tako", "name": "Tako", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "tako", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-08-15", "last_updated": "2024-08-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2048, "output": 0 } }, "xai/grok-code-fast-1": { "id": "xai/grok-code-fast-1", "name": "Grok Code Fast 1", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-08-22", "last_updated": "2025-08-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.5, "cache_read": 0.02 } }, "xai/grok-4.1-fast-reasoning": { "id": "xai/grok-4.1-fast-reasoning", "name": "Grok-4.1-Fast-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 } }, "xai/grok-3-mini": { "id": "xai/grok-3-mini", "name": "Grok 3 Mini", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-04-11", "last_updated": "2025-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.3, "output": 0.5, "cache_read": 0.075 } }, "xai/grok-4.1-fast-non-reasoning": { "id": "xai/grok-4.1-fast-non-reasoning", "name": "Grok-4.1-Fast-Non-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 } }, "xai/grok-3": { "id": "xai/grok-3", "name": "Grok 3", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-04-11", "last_updated": "2025-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75 } }, "xai/grok-4-fast-reasoning": { "id": "xai/grok-4-fast-reasoning", "name": "Grok-4-Fast-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-09-16", "last_updated": "2025-09-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 128000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "xai/grok-4": { "id": "xai/grok-4", "name": "Grok-4", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75 } }, "xai/grok-4-fast-non-reasoning": { "id": "xai/grok-4-fast-non-reasoning", "name": "Grok-4-Fast-Non-Reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-09-16", "last_updated": "2025-09-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 128000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "xai/grok-4.20-multi-agent": { "id": "xai/grok-4.20-multi-agent", "name": "Grok-4.20-Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2026-03-13", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2 } }, "topazlabs-co/topazlabs": { "id": "topazlabs-co/topazlabs", "name": "TopazLabs", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "topazlabs", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 204, "output": 0 } }, "fireworks-ai/kimi-k2.5-fw": { "id": "fireworks-ai/kimi-k2.5-fw", "name": "Kimi-K2.5-FW", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 245760, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "google/veo-3.1-fast": { "id": "google/veo-3.1-fast", "name": "Veo-3.1-Fast", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/imagen-3": { "id": "google/imagen-3", "name": "Imagen-3", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-10-15", "last_updated": "2024-10-15", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/nano-banana-pro": { "id": "google/nano-banana-pro", "name": "Nano-Banana-Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "nano-banana", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 65536, "output": 0 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/lyria": { "id": "google/lyria", "name": "Lyria", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "lyria", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-06-04", "last_updated": "2025-06-04", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Gemini-3.1-Flash-Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2026-02-18", "last_updated": "2026-02-18", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5 } }, "google/nano-banana": { "id": "google/nano-banana", "name": "Nano-Banana", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "nano-banana", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 0 }, "cost": { "input": 0.21, "output": 1.8, "cache_read": 0.021 } }, "google/gemini-deep-research": { "id": "google/gemini-deep-research", "name": "gemini-deep-research", "description": "Legacy model retained for compatibility with older integrations", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 0 }, "status": "deprecated", "cost": { "input": 1.6, "output": 9.6 } }, "google/veo-3": { "id": "google/veo-3", "name": "Veo-3", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-05-21", "last_updated": "2025-05-21", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-3-flash": { "id": "google/gemini-3-flash", "name": "Gemini-3-Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-10-07", "last_updated": "2025-10-07", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.4, "output": 2.4, "cache_read": 0.04 } }, "google/imagen-3-fast": { "id": "google/imagen-3-fast", "name": "Imagen-3-Fast", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-10-17", "last_updated": "2024-10-17", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini-2.5-Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 32768 } ], "tool_call": true, "temperature": false, "release_date": "2025-02-05", "last_updated": "2025-02-05", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1065535, "output": 65535 }, "cost": { "input": 0.87, "output": 7, "cache_read": 0.087 } }, "google/veo-3.1": { "id": "google/veo-3.1", "name": "Veo-3.1", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini-2.5-Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": false, "release_date": "2025-04-26", "last_updated": "2025-04-26", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1065535, "output": 65535 }, "cost": { "input": 0.21, "output": 1.8, "cache_read": 0.021 } }, "google/imagen-4-fast": { "id": "google/imagen-4-fast", "name": "Imagen-4-Fast", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-06-25", "last_updated": "2025-06-25", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini-3.5-Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5152, "output": 9.0909, "cache_read": 0.1515 } }, "google/gemma-4-31b": { "id": "google/gemma-4-31b", "name": "Gemma-4-31B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "google/veo-3-fast": { "id": "google/veo-3-fast", "name": "Veo-3-Fast", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-10-13", "last_updated": "2025-10-13", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-3-pro": { "id": "google/gemini-3-pro", "name": "Gemini-3-Pro", "description": "Legacy model retained for compatibility with older integrations", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-10-22", "last_updated": "2025-10-22", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "status": "deprecated", "cost": { "input": 1.6, "output": 9.6, "cache_read": 0.16 } }, "google/gemini-2.0-flash": { "id": "google/gemini-2.0-flash", "name": "Gemini-2.0-Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 990000, "output": 8192 }, "cost": { "input": 0.1, "output": 0.42 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini-2.5-Flash-Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": false, "release_date": "2025-06-19", "last_updated": "2025-06-19", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1024000, "output": 64000 }, "cost": { "input": 0.07, "output": 0.28 } }, "google/imagen-4": { "id": "google/imagen-4", "name": "Imagen-4", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/imagen-4-ultra": { "id": "google/imagen-4-ultra", "name": "Imagen-4-Ultra", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-05-24", "last_updated": "2025-05-24", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-3.1-pro": { "id": "google/gemini-3.1-pro", "name": "Gemini-3.1-Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/veo-2": { "id": "google/veo-2", "name": "Veo-2", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-12-02", "last_updated": "2024-12-02", "modalities": { "input": ["text"], "output": ["video"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-2.0-flash-lite": { "id": "google/gemini-2.0-flash-lite", "name": "Gemini-2.0-Flash-Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-02-05", "last_updated": "2025-02-05", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 990000, "output": 8192 }, "cost": { "input": 0.052, "output": 0.21 } }, "elevenlabs/elevenlabs-music": { "id": "elevenlabs/elevenlabs-music", "name": "ElevenLabs-Music", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "elevenlabs", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-08-29", "last_updated": "2025-08-29", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 2000, "output": 0 } }, "elevenlabs/elevenlabs-v2.5-turbo": { "id": "elevenlabs/elevenlabs-v2.5-turbo", "name": "ElevenLabs-v2.5-Turbo", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "elevenlabs", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-10-28", "last_updated": "2024-10-28", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 } }, "elevenlabs/elevenlabs-v3": { "id": "elevenlabs/elevenlabs-v3", "name": "ElevenLabs-v3", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "elevenlabs", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 } }, "openai/chatgpt-4o-latest": { "id": "openai/chatgpt-4o-latest", "name": "ChatGPT-4o-Latest", "description": "Legacy model retained for compatibility with older integrations", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-08-14", "last_updated": "2024-08-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "status": "deprecated", "cost": { "input": 4.5, "output": 14 } }, "openai/gpt-3.5-turbo-instruct": { "id": "openai/gpt-3.5-turbo-instruct", "name": "GPT-3.5-Turbo-Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2023-09-20", "last_updated": "2023-09-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 3500, "output": 1024 }, "cost": { "input": 1.4, "output": 1.8 } }, "openai/o3": { "id": "openai/o3", "name": "o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.8, "output": 7.2, "cache_read": 0.45 } }, "openai/gpt-5.2-instant": { "id": "openai/gpt-5.2-instant", "name": "GPT-5.2-Instant", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.6, "output": 13, "cache_read": 0.16 } }, "openai/gpt-4o-search": { "id": "openai/gpt-4o-search", "name": "GPT-4o-Search", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-03-11", "last_updated": "2025-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 2.2, "output": 9 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "GPT-5.2-Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 19, "output": 150 } }, "openai/gpt-4-classic-0314": { "id": "openai/gpt-4-classic-0314", "name": "GPT-4-Classic-0314", "description": "Legacy model retained for compatibility with older integrations", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-08-26", "last_updated": "2024-08-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "status": "deprecated", "cost": { "input": 27, "output": 54 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.1, "output": 9, "cache_read": 0.11 } }, "openai/gpt-5-chat": { "id": "openai/gpt-5-chat", "name": "GPT-5-Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.1, "output": 9, "cache_read": 0.11 } }, "openai/gpt-3.5-turbo": { "id": "openai/gpt-3.5-turbo", "name": "GPT-3.5-Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2023-09-13", "last_updated": "2023-09-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 2048 }, "cost": { "input": 0.45, "output": 1.4 } }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", "name": "GPT-5-Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 14, "output": 110 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 0.99, "output": 4, "cache_read": 0.25 } }, "openai/sora-2": { "id": "openai/sora-2", "name": "Sora-2", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "sora", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "openai/o3-pro": { "id": "openai/o3-pro", "name": "o3-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 18, "output": 72 } }, "openai/gpt-4-classic": { "id": "openai/gpt-4-classic", "name": "GPT-4-Classic", "description": "Legacy model retained for compatibility with older integrations", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-03-25", "last_updated": "2024-03-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 4096 }, "status": "deprecated", "cost": { "input": 27, "output": 54 } }, "openai/gpt-4o-aug": { "id": "openai/gpt-4o-aug", "name": "GPT-4o-Aug", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-11-21", "last_updated": "2024-11-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 2.2, "output": 9, "cache_read": 1.1 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4-Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.18, "output": 1.1, "cache_read": 0.018 } }, "openai/sora-2-pro": { "id": "openai/sora-2-pro", "name": "Sora-2-Pro", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "sora", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT-5.1-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-11-12", "last_updated": "2025-11-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.1, "output": 9, "cache_read": 0.11 } }, "openai/gpt-5.3-instant": { "id": "openai/gpt-5.3-instant", "name": "GPT-5.3-Instant", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 111616, "output": 16384 }, "cost": { "input": 1.6, "output": 13, "cache_read": 0.16 } }, "openai/gpt-5.3-codex-spark": { "id": "openai/gpt-5.3-codex-spark", "name": "GPT-5.3-Codex-Spark", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2026-03-04", "last_updated": "2026-03-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-5.1-codex-max": { "id": "openai/gpt-5.1-codex-max", "name": "GPT-5.1-Codex-Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.1, "output": 9, "cache_read": 0.11 } }, "openai/dall-e-3": { "id": "openai/dall-e-3", "name": "DALL-E-3", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "dall-e", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2023-11-06", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 800, "output": 0 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "o3-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 0.99, "output": 4 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.6, "output": 13, "cache_read": 0.16 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT-5.3-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-02-10", "last_updated": "2026-02-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.6, "output": 13, "cache_read": 0.16 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "GPT-5.1-Codex-Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-11-12", "last_updated": "2025-11-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.22, "output": 1.8, "cache_read": 0.022 } }, "openai/o4-mini-deep-research": { "id": "openai/o4-mini-deep-research", "name": "o4-mini-deep-research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-06-27", "last_updated": "2025-06-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.8, "output": 7.2, "cache_read": 0.45 } }, "openai/gpt-image-1.5": { "id": "openai/gpt-image-1.5", "name": "gpt-image-1.5", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT-4.1-nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.09, "output": 0.36, "cache_read": 0.022 } }, "openai/gpt-3.5-turbo-raw": { "id": "openai/gpt-3.5-turbo-raw", "name": "GPT-3.5-Turbo-Raw", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2023-09-27", "last_updated": "2023-09-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4524, "output": 2048 }, "cost": { "input": 0.45, "output": 1.4 } }, "openai/o1": { "id": "openai/o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2024-12-18", "last_updated": "2024-12-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 14, "output": 54 } }, "openai/o1-pro": { "id": "openai/o1-pro", "name": "o1-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-03-19", "last_updated": "2025-03-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 140, "output": 540 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "pdf"], "output": ["image"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.2, "output": 14, "cache_read": 0.22 } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4-Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-03-12", "last_updated": "2026-03-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.68, "output": 4, "cache_read": 0.068 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 1.8, "output": 7.2, "cache_read": 0.45 } }, "openai/o3-deep-research": { "id": "openai/o3-deep-research", "name": "o3-deep-research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-06-27", "last_updated": "2025-06-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 9, "output": 36, "cache_read": 2.2 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-06-25", "last_updated": "2025-06-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.22, "output": 1.8, "cache_read": 0.022 } }, "openai/gpt-image-1": { "id": "openai/gpt-image-1", "name": "GPT-Image-1", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-03-31", "last_updated": "2025-03-31", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.36, "output": 1.4, "cache_read": 0.09 } }, "openai/gpt-4-turbo": { "id": "openai/gpt-4-turbo", "name": "GPT-4-Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2023-09-13", "last_updated": "2023-09-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 9, "output": 27 } }, "openai/gpt-image-1-mini": { "id": "openai/gpt-image-1-mini", "name": "GPT-Image-1-Mini", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5-nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.045, "output": 0.36, "cache_read": 0.0045 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "GPT-5.4-Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 27, "output": 160 } }, "openai/o3-mini-high": { "id": "openai/o3-mini-high", "name": "o3-mini-high", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 0.99, "output": 4 } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT-5.5-Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-08", "last_updated": "2026-04-08", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 27.2727, "output": 163.6364 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 124096, "output": 4096 }, "cost": { "input": 0.14, "output": 0.54, "cache_read": 0.068 } }, "openai/gpt-4o-mini-search": { "id": "openai/gpt-4o-mini-search", "name": "GPT-4o-mini-Search", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-03-11", "last_updated": "2025-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.14, "output": 0.54 } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.1, "output": 9 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT-5.2-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.6, "output": 13, "cache_read": 0.16 } }, "openai/gpt-5.1-instant": { "id": "openai/gpt-5.1-instant", "name": "GPT-5.1-Instant", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-11-12", "last_updated": "2025-11-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.1, "output": 9, "cache_read": 0.11 } }, "openai/gpt-image-2": { "id": "openai/gpt-image-2", "name": "GPT-Image-2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "cost": { "input": 5.0505, "output": 32.3232, "cache_read": 1.2626 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-11-12", "last_updated": "2025-11-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.1, "output": 9, "cache_read": 0.11 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-08", "last_updated": "2026-04-08", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 4.5455, "output": 27.2727, "cache_read": 0.4545 } }, "runwayml/runway": { "id": "runwayml/runway", "name": "Runway", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "runway", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-10-11", "last_updated": "2024-10-11", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 256, "output": 0 } }, "runwayml/runway-gen-4-turbo": { "id": "runwayml/runway-gen-4-turbo", "name": "Runway-Gen-4-Turbo", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "runway", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-05-09", "last_updated": "2025-05-09", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 256, "output": 0 } }, "cerebras/llama-3.1-8b-cs": { "id": "cerebras/llama-3.1-8b-cs", "name": "Llama-3.1-8B-CS", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-05-13", "last_updated": "2025-05-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 }, "cost": { "input": 0.1, "output": 0.1 } }, "cerebras/qwen3-235b-2507-cs": { "id": "cerebras/qwen3-235b-2507-cs", "name": "qwen3-235b-2507-cs", "description": "Legacy model retained for compatibility with older integrations", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-08-06", "last_updated": "2025-08-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "status": "deprecated" }, "cerebras/gpt-oss-120b-cs": { "id": "cerebras/gpt-oss-120b-cs", "name": "GPT-OSS-120B-CS", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-08-06", "last_updated": "2025-08-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 }, "cost": { "input": 0.35, "output": 0.75 } }, "cerebras/llama-3.3-70b-cs": { "id": "cerebras/llama-3.3-70b-cs", "name": "llama-3.3-70b-cs", "description": "Legacy model retained for compatibility with older integrations", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-13", "last_updated": "2025-05-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "status": "deprecated" }, "cerebras/qwen3-32b-cs": { "id": "cerebras/qwen3-32b-cs", "name": "qwen3-32b-cs", "description": "Legacy model retained for compatibility with older integrations", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-05-15", "last_updated": "2025-05-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 }, "status": "deprecated" }, "anthropic/claude-sonnet-3.5": { "id": "anthropic/claude-sonnet-3.5", "name": "Claude-Sonnet-3.5", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-06-05", "last_updated": "2024-06-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 189096, "output": 8192 }, "status": "deprecated", "cost": { "input": 2.6, "output": 13, "cache_read": 0.26, "cache_write": 3.2 } }, "anthropic/claude-sonnet-3.7": { "id": "anthropic/claude-sonnet-3.7", "name": "Claude-Sonnet-3.7", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 196608, "output": 128000 }, "cost": { "input": 2.6, "output": 13, "cache_read": 0.26, "cache_write": 3.2 } }, "anthropic/claude-sonnet-3.5-june": { "id": "anthropic/claude-sonnet-3.5-june", "name": "Claude-Sonnet-3.5-June", "description": "Legacy model retained for compatibility with older integrations", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-11-18", "last_updated": "2024-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 189096, "output": 8192 }, "status": "deprecated", "cost": { "input": 2.6, "output": 13, "cache_read": 0.26, "cache_write": 3.2 } }, "anthropic/claude-sonnet-4.5": { "id": "anthropic/claude-sonnet-4.5", "name": "Claude-Sonnet-4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 31999 } ], "tool_call": true, "temperature": false, "release_date": "2025-09-26", "last_updated": "2025-09-26", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983040, "output": 32768 }, "cost": { "input": 2.6, "output": 13, "cache_read": 0.26, "cache_write": 3.2 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude-Sonnet-4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-05-21", "last_updated": "2025-05-21", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983040, "output": 64000 }, "cost": { "input": 2.6, "output": 13, "cache_read": 0.26, "cache_write": 3.2 } }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude-Haiku-4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 63999 } ], "tool_call": true, "temperature": false, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 192000, "output": 64000 }, "cost": { "input": 0.85, "output": 4.3, "cache_read": 0.085, "cache_write": 1.1 } }, "anthropic/claude-haiku-3": { "id": "anthropic/claude-haiku-3", "name": "Claude-Haiku-3", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-03-09", "last_updated": "2024-03-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 189096, "output": 8192 }, "cost": { "input": 0.21, "output": 1.1, "cache_read": 0.021, "cache_write": 0.26 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude-Opus-4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-15", "last_updated": "2026-04-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 128000 }, "cost": { "input": 4.3, "output": 21, "cache_read": 0.43, "cache_write": 5.4 } }, "anthropic/claude-haiku-3.5": { "id": "anthropic/claude-haiku-3.5", "name": "Claude-Haiku-3.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 189096, "output": 8192 }, "cost": { "input": 0.68, "output": 3.4, "cache_read": 0.068, "cache_write": 0.85 } }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude-Opus-4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 128000 }, "cost": { "input": 4.2929, "output": 21.4646 } }, "anthropic/claude-opus-4.1": { "id": "anthropic/claude-opus-4.1", "name": "Claude-Opus-4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 31999 } ], "tool_call": true, "temperature": false, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 196608, "output": 32000 }, "cost": { "input": 13, "output": 64, "cache_read": 1.3, "cache_write": 16 } }, "anthropic/claude-opus-4.5": { "id": "anthropic/claude-opus-4.5", "name": "Claude-Opus-4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 0, "max": 63999 } ], "tool_call": true, "temperature": false, "release_date": "2025-11-21", "last_updated": "2025-11-21", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 196608, "output": 64000 }, "cost": { "input": 4.3, "output": 21, "cache_read": 0.43, "cache_write": 5.3 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude-Sonnet-4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "temperature": false, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983040, "output": 128000 }, "cost": { "input": 2.6, "output": 13, "cache_read": 0.26, "cache_write": 3.2 } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Claude-Opus-4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-05-21", "last_updated": "2025-05-21", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 192512, "output": 28672 }, "cost": { "input": 13, "output": 64, "cache_read": 1.3, "cache_write": 16 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Claude-Opus-4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "temperature": false, "release_date": "2026-02-04", "last_updated": "2026-02-04", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983040, "output": 128000 }, "cost": { "input": 4.3, "output": 21, "cache_read": 0.43, "cache_write": 5.3 } }, "lumalabs/ray2": { "id": "lumalabs/ray2", "name": "Ray2", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "ray", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-02-20", "last_updated": "2025-02-20", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 5000, "output": 0 } }, "poetools/claude-code": { "id": "poetools/claude-code", "name": "claude-code", "description": "Claude model for careful reasoning, writing, coding, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2025-11-27", "last_updated": "2025-11-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "empiriolabs/deepseek-v4-pro-el": { "id": "empiriolabs/deepseek-v4-pro-el", "name": "DeepSeek-V4-Pro-EL", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "release_date": "2026-04-24", "last_updated": "2026-05-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "input": 1000000, "output": 384000 }, "cost": { "input": 1.67, "output": 3.33 } }, "empiriolabs/deepseek-v4-flash-el": { "id": "empiriolabs/deepseek-v4-flash-el", "name": "DeepSeek-V4-Flash-EL", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "release_date": "2026-04-24", "last_updated": "2026-05-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "input": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28 } }, "stabilityai/stablediffusionxl": { "id": "stabilityai/stablediffusionxl", "name": "StableDiffusionXL", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "stable-diffusion", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2023-07-09", "last_updated": "2023-07-09", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 200, "output": 0 } }, "ideogramai/ideogram": { "id": "ideogramai/ideogram", "name": "Ideogram", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "ideogram", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-04-03", "last_updated": "2024-04-03", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 150, "output": 0 } }, "ideogramai/ideogram-v2a": { "id": "ideogramai/ideogram-v2a", "name": "Ideogram-v2a", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "ideogram", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 150, "output": 0 } }, "ideogramai/ideogram-v2": { "id": "ideogramai/ideogram-v2", "name": "Ideogram-v2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "ideogram", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-08-21", "last_updated": "2024-08-21", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 150, "output": 0 } }, "ideogramai/ideogram-v2a-turbo": { "id": "ideogramai/ideogram-v2a-turbo", "name": "Ideogram-v2a-Turbo", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "ideogram", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 150, "output": 0 } }, "novita/glm-4.7": { "id": "novita/glm-4.7", "name": "glm-4.7", "description": "Legacy model retained for compatibility with older integrations", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 205000, "output": 131072 }, "status": "deprecated" }, "novita/minimax-m2.1": { "id": "novita/minimax-m2.1", "name": "minimax-m2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": false, "release_date": "2025-12-26", "last_updated": "2025-12-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 205000, "output": 131072 } }, "novita/glm-4.6": { "id": "novita/glm-4.6", "name": "GLM-4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 0, "output": 0 } }, "novita/kimi-k2-thinking": { "id": "novita/kimi-k2-thinking", "name": "kimi-k2-thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": false, "release_date": "2025-11-07", "last_updated": "2025-11-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 0 } }, "novita/kimi-k2.5": { "id": "novita/kimi-k2.5", "name": "Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "novita/glm-4.6v": { "id": "novita/glm-4.6v", "name": "glm-4.6v", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": false, "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 32768 } }, "novita/kimi-k2.6": { "id": "novita/kimi-k2.6", "name": "Kimi-K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "input": 262144, "output": 262144 }, "cost": { "input": 0.96, "output": 4.04, "cache_read": 0.16 } }, "novita/glm-4.7-flash": { "id": "novita/glm-4.7-flash", "name": "glm-4.7-flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": false, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 65500 } }, "novita/glm-4.7-n": { "id": "novita/glm-4.7-n", "name": "glm-4.7-n", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": false, "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 205000, "output": 131072 } }, "novita/glm-5": { "id": "novita/glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 205000, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "novita/deepseek-v3.2": { "id": "novita/deepseek-v3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 0 }, "cost": { "input": 0.27, "output": 0.4, "cache_read": 0.13 } } } }, "kimi-for-coding": { "id": "kimi-for-coding", "env": ["KIMI_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://api.kimi.com/coding/v1", "name": "Kimi For Coding", "doc": "https://www.kimi.com/code/docs/en/third-party-tools/other-coding-agents.html", "models": { "k2p7": { "id": "k2p7", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-11", "last_updated": "2025-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "k2p5": { "id": "k2p5", "name": "Kimi K2.5", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "k2p6": { "id": "k2p6", "name": "Kimi K2.6", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04", "last_updated": "2026-04", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "dinference": { "id": "dinference", "env": ["DINFERENCE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.dinference.com/v1", "name": "DInference", "doc": "https://dinference.com", "models": { "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.22, "output": 0.88 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.45, "output": 1.65 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 1.25, "output": 3.89 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08", "last_updated": "2025-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.0675, "output": 0.27 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.75, "output": 2.4 } } } }, "perplexity-agent": { "id": "perplexity-agent", "env": ["PERPLEXITY_API_KEY"], "npm": "@ai-sdk/openai", "api": "https://api.perplexity.ai/v1", "name": "Perplexity Agent", "doc": "https://docs.perplexity.ai/docs/agent-api/models", "models": { "xai/grok-4-1-fast-non-reasoning": { "id": "xai/grok-4-1-fast-non-reasoning", "name": "Grok 4.1 Fast (Non-Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-11-19", "last_updated": "2025-11-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-03-20", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-03-20", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "tiers": [ { "input": 0.5, "output": 3, "cache_read": 0.05, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 0.5, "output": 3, "cache_read": 0.05 } } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nemotron 3 Super 120B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2026-02", "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 0.25, "output": 2.5 } }, "anthropic/claude-opus-4-5": { "id": "anthropic/claude-opus-4-5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5 } }, "anthropic/claude-sonnet-4-5": { "id": "anthropic/claude-sonnet-4-5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3 } }, "anthropic/claude-opus-4-7": { "id": "anthropic/claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5 } }, "anthropic/claude-haiku-4-5": { "id": "anthropic/claude-haiku-4-5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1 } }, "anthropic/claude-opus-4-6": { "id": "anthropic/claude-opus-4-6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5 } }, "anthropic/claude-sonnet-4-6": { "id": "anthropic/claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3 } }, "perplexity/sonar": { "id": "perplexity/sonar", "name": "Sonar", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.25, "output": 2.5, "cache_read": 0.0625 } } } }, "siliconflow": { "id": "siliconflow", "env": ["SILICONFLOW_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.siliconflow.com/v1", "name": "SiliconFlow", "doc": "https://cloud.siliconflow.com/models", "models": { "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "moonshotai/Kimi-K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-06-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.77, "output": 4, "cache_read": 0.2 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "moonshotai/Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.45, "output": 2.25 } }, "baidu/ERNIE-4.5-300B-A47B": { "id": "baidu/ERNIE-4.5-300B-A47B", "name": "baidu/ERNIE-4.5-300B-A47B", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "ernie", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-02", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.28, "output": 1.1 } }, "ByteDance-Seed/Seed-OSS-36B-Instruct": { "id": "ByteDance-Seed/Seed-OSS-36B-Instruct", "name": "ByteDance-Seed/Seed-OSS-36B-Instruct", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.21, "output": 0.57 } }, "stepfun-ai/Step-3.5-Flash": { "id": "stepfun-ai/Step-3.5-Flash", "name": "stepfun-ai/Step-3.5-Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "family": "step", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.1, "output": 0.3 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.13, "output": 0.4 } }, "google/gemma-4-26B-A4B-it": { "id": "google/gemma-4-26B-A4B-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.12, "output": 0.4 } }, "inclusionAI/Ling-flash-2.0": { "id": "inclusionAI/Ling-flash-2.0", "name": "inclusionAI/Ling-flash-2.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-18", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.57 } }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2, "output": 1.6 } }, "Qwen/Qwen2.5-7B-Instruct": { "id": "Qwen/Qwen2.5-7B-Instruct", "name": "Qwen/Qwen2.5-7B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 33000, "output": 4000 }, "cost": { "input": 0.05, "output": 0.05 } }, "Qwen/Qwen3-VL-235B-A22B-Instruct": { "id": "Qwen/Qwen3-VL-235B-A22B-Instruct", "name": "Qwen/Qwen3-VL-235B-A22B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-04", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.3, "output": 1.5 } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.3, "output": 3.2 } }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.39, "output": 2.34 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen/Qwen3-235B-A22B-Thinking-2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.13, "output": 0.6 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-31", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.25, "output": 1 } }, "Qwen/Qwen3.5-122B-A10B": { "id": "Qwen/Qwen3.5-122B-A10B", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.26, "output": 2.08 } }, "Qwen/Qwen3-Coder-30B-A3B-Instruct": { "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "name": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-01", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.07, "output": 0.28 } }, "Qwen/Qwen3.5-27B": { "id": "Qwen/Qwen3.5-27B", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.25, "output": 2 } }, "Qwen/Qwen3-30B-A3B-Instruct-2507": { "id": "Qwen/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen/Qwen3-30B-A3B-Instruct-2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.09, "output": 0.3 } }, "Qwen/Qwen3-VL-30B-A3B-Thinking": { "id": "Qwen/Qwen3-VL-30B-A3B-Thinking", "name": "Qwen/Qwen3-VL-30B-A3B-Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-11", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.29, "output": 1 } }, "Qwen/Qwen3-VL-8B-Instruct": { "id": "Qwen/Qwen3-VL-8B-Instruct", "name": "Qwen/Qwen3-VL-8B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-15", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.18, "output": 0.68 } }, "Qwen/Qwen3-VL-235B-A22B-Thinking": { "id": "Qwen/Qwen3-VL-235B-A22B-Thinking", "name": "Qwen/Qwen3-VL-235B-A22B-Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-04", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.45, "output": 3.5 } }, "Qwen/Qwen3-8B": { "id": "Qwen/Qwen3-8B", "name": "Qwen/Qwen3-8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.06, "output": 0.06 } }, "Qwen/Qwen3.5-9B": { "id": "Qwen/Qwen3.5-9B", "name": "Qwen/Qwen3.5-9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.1, "output": 0.15 } }, "Qwen/Qwen3-32B": { "id": "Qwen/Qwen3-32B", "name": "Qwen/Qwen3-32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.57 } }, "Qwen/Qwen3-VL-32B-Instruct": { "id": "Qwen/Qwen3-VL-32B-Instruct", "name": "Qwen/Qwen3-VL-32B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-21", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.2, "output": 0.6 } }, "Qwen/Qwen3-14B": { "id": "Qwen/Qwen3-14B", "name": "Qwen/Qwen3-14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.07, "output": 0.28 } }, "Qwen/Qwen2.5-72B-Instruct": { "id": "Qwen/Qwen2.5-72B-Instruct", "name": "Qwen/Qwen2.5-72B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 33000, "output": 4000 }, "cost": { "input": 0.59, "output": 0.59 } }, "Qwen/Qwen3-VL-30B-A3B-Instruct": { "id": "Qwen/Qwen3-VL-30B-A3B-Instruct", "name": "Qwen/Qwen3-VL-30B-A3B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-05", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.29, "output": 1 } }, "Qwen/Qwen3-VL-32B-Thinking": { "id": "Qwen/Qwen3-VL-32B-Thinking", "name": "Qwen/Qwen3-VL-32B-Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-21", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.2, "output": 1.5 } }, "Qwen/Qwen3.5-35B-A3B": { "id": "Qwen/Qwen3.5-35B-A3B", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.24, "output": 1.8 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "openai/gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-13", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 8000 }, "cost": { "input": 0.05, "output": 0.45 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "openai/gpt-oss-20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-13", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 8000 }, "cost": { "input": 0.04, "output": 0.18 } }, "tencent/Hy3-preview": { "id": "tencent/Hy3-preview", "name": "Hy3 preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.066, "output": 0.26, "cache_read": 0.029 } }, "tencent/Hunyuan-A13B-Instruct": { "id": "tencent/Hunyuan-A13B-Instruct", "name": "tencent/Hunyuan-A13B-Instruct", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.57 } }, "zai-org/GLM-5V-Turbo": { "id": "zai-org/GLM-5V-Turbo", "name": "zai-org/GLM-5V-Turbo", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_write": 0 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "zai-org/GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 205000, "output": 205000 }, "cost": { "input": 0.95, "output": 2.55 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1049000, "output": 262000 }, "cost": { "input": 1.4, "output": 4.4, "cache_write": 0 } }, "zai-org/GLM-4.5-Air": { "id": "zai-org/GLM-4.5-Air", "name": "zai-org/GLM-4.5-Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.14, "output": 0.86 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "zai-org/GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-04-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 205000, "output": 205000 }, "cost": { "input": 1.4, "output": 4.4, "cache_write": 0 } }, "deepseek-ai/DeepSeek-R1": { "id": "deepseek-ai/DeepSeek-R1", "name": "deepseek-ai/DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-05-28", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.5, "output": 2.18 } }, "deepseek-ai/DeepSeek-V3.1-Terminus": { "id": "deepseek-ai/DeepSeek-V3.1-Terminus", "name": "deepseek-ai/DeepSeek-V3.1-Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-29", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 1 } }, "deepseek-ai/DeepSeek-V3.1": { "id": "deepseek-ai/DeepSeek-V3.1", "name": "deepseek-ai/DeepSeek-V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-25", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 1 } }, "deepseek-ai/DeepSeek-V3.2-Exp": { "id": "deepseek-ai/DeepSeek-V3.2-Exp", "name": "deepseek-ai/DeepSeek-V3.2-Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-10", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 0.41 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.145 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "deepseek-ai/DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 0.42 } }, "deepseek-ai/DeepSeek-V3": { "id": "deepseek-ai/DeepSeek-V3", "name": "deepseek-ai/DeepSeek-V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-26", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.25, "output": 1 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMaxAI/MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 197000, "output": 131000 }, "cost": { "input": 0.3, "output": 1.2 } } } }, "umans-ai-coding-plan": { "id": "umans-ai-coding-plan", "env": ["UMANS_AI_CODING_PLAN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.code.umans.ai/v1", "name": "Umans AI Coding Plan", "doc": "https://app.umans.ai/offers/code/docs", "models": { "umans-kimi-k2.7": { "id": "umans-kimi-k2.7", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "umans-glm-5.1": { "id": "umans-glm-5.1", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "umans-coder": { "id": "umans-coder", "name": "Umans Coder", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "umans-flash": { "id": "umans-flash", "name": "Umans Flash", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "umans-glm-5.2": { "id": "umans-glm-5.2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 405504, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "umans-qwen3.6-35b-a3b": { "id": "umans-qwen3.6-35b-a3b", "name": "Qwen3.6 35B A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "io-net": { "id": "io-net", "env": ["IOINTELLIGENCE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.intelligence.io.solutions/api/v1", "name": "IO.NET", "doc": "https://io.net/docs/guides/intelligence/io-intelligence", "models": { "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "name": "Llama 4 Maverick 17B 128E Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-15", "last_updated": "2025-01-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 430000, "output": 4096 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075, "cache_write": 0.3 } }, "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.13, "output": 0.38, "cache_read": 0.065, "cache_write": 0.26 } }, "meta-llama/Llama-3.2-90B-Vision-Instruct": { "id": "meta-llama/Llama-3.2-90B-Vision-Instruct", "name": "Llama 3.2 90B Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 4096 }, "cost": { "input": 0.35, "output": 0.4, "cache_read": 0.175, "cache_write": 0.7 } }, "moonshotai/Kimi-K2-Thinking": { "id": "moonshotai/Kimi-K2-Thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.55, "output": 2.25, "cache_read": 0.275, "cache_write": 1.1 } }, "moonshotai/Kimi-K2-Instruct-0905": { "id": "moonshotai/Kimi-K2-Instruct-0905", "name": "Kimi K2 Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2024-09-05", "last_updated": "2024-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.39, "output": 1.9, "cache_read": 0.195, "cache_write": 0.78 } }, "Qwen/Qwen2.5-VL-32B-Instruct": { "id": "Qwen/Qwen2.5-VL-32B-Instruct", "name": "Qwen 2.5 VL 32B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.05, "output": 0.22, "cache_read": 0.025, "cache_write": 0.1 } }, "Qwen/Qwen3-Next-80B-A3B-Instruct": { "id": "Qwen/Qwen3-Next-80B-A3B-Instruct", "name": "Qwen 3 Next 80B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-10", "last_updated": "2025-01-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 4096 }, "cost": { "input": 0.1, "output": 0.8, "cache_read": 0.05, "cache_write": 0.2 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen 3 235B Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 4096 }, "cost": { "input": 0.11, "output": 0.6, "cache_read": 0.055, "cache_write": 0.22 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT-OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 4096 }, "cost": { "input": 0.04, "output": 0.4, "cache_read": 0.02, "cache_write": 0.08 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT-OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 4096 }, "cost": { "input": 0.03, "output": 0.14, "cache_read": 0.015, "cache_write": 0.06 } }, "mistralai/Devstral-Small-2505": { "id": "mistralai/Devstral-Small-2505", "name": "Devstral Small 2505", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-05-01", "last_updated": "2025-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.05, "output": 0.22, "cache_read": 0.025, "cache_write": 0.1 } }, "mistralai/Magistral-Small-2506": { "id": "mistralai/Magistral-Small-2506", "name": "Magistral Small 2506", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-small", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 0.25, "cache_write": 1 } }, "mistralai/Mistral-Large-Instruct-2411": { "id": "mistralai/Mistral-Large-Instruct-2411", "name": "Mistral Large Instruct 2411", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 2, "output": 6, "cache_read": 1, "cache_write": 4 } }, "mistralai/Mistral-Nemo-Instruct-2407": { "id": "mistralai/Mistral-Nemo-Instruct-2407", "name": "Mistral Nemo Instruct 2407", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-05", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.02, "output": 0.04, "cache_read": 0.01, "cache_write": 0.04 } }, "Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar": { "id": "Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar", "name": "Qwen 3 Coder 480B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-15", "last_updated": "2025-01-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 106000, "output": 4096 }, "cost": { "input": 0.22, "output": 0.95, "cache_read": 0.11, "cache_write": 0.44 } }, "zai-org/GLM-4.6": { "id": "zai-org/GLM-4.6", "name": "GLM 4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-11-15", "last_updated": "2024-11-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 0.4, "output": 1.75, "cache_read": 0.2, "cache_write": 0.8 } }, "deepseek-ai/DeepSeek-R1-0528": { "id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 2, "output": 8.75, "cache_read": 1, "cache_write": 4 } } } }, "gmicloud": { "id": "gmicloud", "env": ["GMICLOUD_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.gmi-serving.com/v1", "name": "GMI Cloud", "doc": "https://docs.gmicloud.ai/inference-engine/api-reference/llm-api-reference", "models": { "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.855, "output": 3.6, "cache_read": 0.144 } }, "moonshotai/kimi-k2.7-code-highspeed": { "id": "moonshotai/kimi-k2.7-code-highspeed", "name": "Kimi K2.7 Code Highspeed", "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.9, "output": 8, "cache_read": 0.38 } }, "Qwen/Qwen3.7-Max": { "id": "Qwen/Qwen3.7-Max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.25, "cache_write": 3.125 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 409600, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 4.5, "output": 22.5, "cache_read": 0.45 } }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 409600, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 409600, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5 } }, "zai-org/GLM-5.1-FP8": { "id": "zai-org/GLM-5.1-FP8", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.98, "output": 3.08, "cache_read": 0.182 } }, "zai-org/GLM-5-FP8": { "id": "zai-org/GLM-5-FP8", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.6, "output": 1.92, "cache_read": 0.12 } }, "zai-org/GLM-5.2-FP8": { "id": "zai-org/GLM-5.2-FP8", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0.979, "output": 3.08, "cache_read": 0.182 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048575, "output": 384000 }, "cost": { "input": 0.112, "output": 0.224, "cache_read": 0.022 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 1.392, "output": 2.784, "cache_read": 0.116 } } } }, "xiaomi-token-plan-cn": { "id": "xiaomi-token-plan-cn", "env": ["XIAOMI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://token-plan-cn.xiaomimimo.com/v1", "name": "Xiaomi Token Plan (China)", "doc": "https://platform.xiaomimimo.com/#/docs", "models": { "mimo-v2.5-tts-voiceclone": { "id": "mimo-v2.5-tts-voiceclone", "name": "MiMo-V2.5-TTS-VoiceClone", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5-tts-voicedesign": { "id": "mimo-v2.5-tts-voicedesign", "name": "MiMo-V2.5-TTS-VoiceDesign", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5": { "id": "mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2-tts": { "id": "mimo-v2-tts", "name": "MiMo-V2-TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2-pro": { "id": "mimo-v2-pro", "name": "MiMo-V2-Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 131072 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2.5-pro": { "id": "mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2.5-tts": { "id": "mimo-v2.5-tts", "name": "MiMo-V2.5-TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } } } }, "zeldoc": { "id": "zeldoc", "env": ["ZELDOC_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.zeldoc.ai/v1", "name": "Zeldoc", "doc": "https://docs.zeldoc.ai", "models": { "z-code": { "id": "z-code", "name": "Z-Code", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-15", "last_updated": "2026-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0 } } } }, "scaleway": { "id": "scaleway", "env": ["SCALEWAY_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.scaleway.ai/v1", "name": "Scaleway", "doc": "https://www.scaleway.com/en/docs/generative-apis/", "models": { "qwen3-235b-a22b-instruct-2507": { "id": "qwen3-235b-a22b-instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-01", "last_updated": "2026-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 260000, "output": 16384 }, "cost": { "input": 0.75, "output": 2.25, "reasoning": 8.4 } }, "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2026-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.2, "output": 0.8 } }, "qwen3-embedding-8b": { "id": "qwen3-embedding-8b", "name": "Qwen3 Embedding 8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-25-11", "last_updated": "2026-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.1, "output": 0 } }, "bge-multilingual-gemma2": { "id": "bge-multilingual-gemma2", "name": "BGE Multilingual Gemma2", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-07-26", "last_updated": "2025-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 3072 }, "cost": { "input": 0.1, "output": 0 } }, "qwen3.6-35b-a3b": { "id": "qwen3.6-35b-a3b", "name": "Qwen3.6 35B A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-05-01", "last_updated": "2026-05-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "status": "beta", "cost": { "input": 0.25, "output": 1.5 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2026-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 100000, "output": 16384 }, "cost": { "input": 0.9, "output": 0.9 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 1.8, "output": 5.5 } }, "pixtral-12b-2409": { "id": "pixtral-12b-2409", "name": "Pixtral 12B 2409", "description": "Mistral vision-language model for image understanding and multimodal chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-09-25", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.2, "output": 0.2 } }, "mistral-small-3.2-24b-instruct-2506": { "id": "mistral-small-3.2-24b-instruct-2506", "name": "Mistral Small 3.2 24B Instruct (2506)", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-06-20", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.35 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT-OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2024-01-01", "last_updated": "2026-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6 } }, "gemma-4-26b-a4b-it": { "id": "gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-01", "last_updated": "2026-05-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "status": "beta", "cost": { "input": 0.25, "output": 0.5 } }, "mistral-medium-3.5-128b": { "id": "mistral-medium-3.5-128b", "name": "Mistral Medium 3.5 128B", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 1.5, "output": 7.5 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen3.5 397B A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "cost": { "input": 0.6, "output": 3.6 } }, "whisper-large-v3": { "id": "whisper-large-v3", "name": "Whisper Large v3", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2023-09", "release_date": "2023-09-01", "last_updated": "2026-03-17", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 0, "output": 8192 }, "cost": { "input": 0.003, "output": 0 } }, "gemma-3-27b-it": { "id": "gemma-3-27b-it", "name": "Gemma-3-27B-IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-01", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 40000, "output": 8192 }, "cost": { "input": 0.25, "output": 0.5 } }, "voxtral-small-24b-2507": { "id": "voxtral-small-24b-2507", "name": "Voxtral Small 24B 2507", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "voxtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-01", "last_updated": "2026-03-17", "modalities": { "input": ["text", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.35 } }, "devstral-2-123b-instruct-2512": { "id": "devstral-2-123b-instruct-2512", "name": "Devstral 2 123B Instruct (2512)", "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-01-07", "last_updated": "2026-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16384 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2 } } } }, "empiriolabs": { "id": "empiriolabs", "env": ["EMPIRIOLABS_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.empiriolabs.ai/v1", "name": "EmpirioLabs AI", "doc": "https://docs.empiriolabs.ai", "models": { "qwen3-5-plus": { "id": "qwen3-5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.36, "output": 2.21, "tiers": [ { "input": 1.08, "output": 6.62, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1.08, "output": 6.62 } } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 393216 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 393216 }, "cost": { "input": 0.14, "output": 0.28 } }, "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 16000 }, "cost": { "input": 0.8939, "output": 3.7131, "cache_read": 0.1788 } }, "qwen3-5-27b": { "id": "qwen3-5-27b", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 80000 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.086, "output": 0.688, "tiers": [ { "input": 0.258, "output": 2.064, "tier": { "type": "context", "size": 128000 } } ] } }, "step-3-7-flash": { "id": "step-3-7-flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 131072 }, "cost": { "input": 0.2, "output": 1.15, "cache_read": 0.04 } }, "qwen3-5-397b-a17b": { "id": "qwen3-5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 80000 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.172, "output": 1.032, "tiers": [ { "input": 0.43, "output": 2.58, "tier": { "type": "context", "size": 128000 } } ] } }, "qwen3-5-35b-a3b": { "id": "qwen3-5-35b-a3b", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 80000 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.057, "output": 0.459, "tiers": [ { "input": 0.229, "output": 1.835, "tier": { "type": "context", "size": 128000 } } ] } }, "qwen3-max": { "id": "qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 1.08, "output": 5.52, "tiers": [ { "input": 2.7, "output": 13.8, "tier": { "type": "context", "size": 128000 } } ] } }, "glm-5-1": { "id": "glm-5-1", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 38912 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202000, "output": 128000 }, "cost": { "input": 0.825, "output": 3.301, "cache_read": 0.165, "tiers": [ { "input": 1.1, "output": 3.851, "cache_read": 0.165, "tier": { "type": "context", "size": 32000 } } ] } }, "qwen3-7-plus": { "id": "qwen3-7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 256000 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.4, "output": 1.6, "tiers": [ { "input": 1.2, "output": 4.8, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1.2, "output": 4.8 } } }, "kimi-k2-7-code": { "id": "kimi-k2-7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 131072 }, "cost": { "input": 0.95, "output": 4 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 393216 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 393216 }, "cost": { "input": 1.65, "output": 3.3 } }, "glm-4-5-flash": { "id": "glm-4-5-flash", "name": "GLM 4.5 Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 98304 }, "cost": { "input": 0, "output": 0 } }, "qwen3-6-flash": { "id": "qwen3-6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 64000 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "tiers": [ { "input": 1, "output": 4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1, "output": 4 } } }, "minimax-m2-7-highspeed": { "id": "minimax-m2-7-highspeed", "name": "MiniMax M2.7 Highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32768 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "minimax-m2-7": { "id": "minimax-m2-7", "name": "MiniMax M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.03 } }, "kimi-k2-7-code-highspeed": { "id": "kimi-k2-7-code-highspeed", "name": "Kimi K2.7 Code Highspeed", "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 131072 }, "cost": { "input": 1.9, "output": 8 } }, "minimax-m3": { "id": "minimax-m3", "name": "MiniMax M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 524288 }, "cost": { "input": 0.225, "output": 0.9, "cache_read": 0.045, "tiers": [ { "input": 0.45, "output": 1.8, "cache_read": 0.045, "tier": { "type": "context", "size": 512000 } } ], "context_over_200k": { "input": 0.45, "output": 1.8, "cache_read": 0.045 } } }, "mimo-v2-5": { "id": "mimo-v2-5", "name": "MiMo V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.7, "output": 1.4, "cache_read": 0.014 } }, "qwen3-6-plus": { "id": "qwen3-6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "tiers": [ { "input": 2, "output": 6, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6 } } }, "step-3-5-flash": { "id": "step-3-5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 131072 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "qwen3-6-27b": { "id": "qwen3-6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 80000 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.412564, "output": 2.475384 } }, "glm-5-2": { "id": "glm-5-2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4 } }, "glm-4-7-flash": { "id": "glm-4-7-flash", "name": "GLM 4.7 Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "step-3-5-flash-2603": { "id": "step-3-5-flash-2603", "name": "Step 3.5 Flash 2603", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 131072 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "qwen3-5-4b": { "id": "qwen3-5-4b", "name": "Qwen3.5 4B", "description": "Qwen3.5 4B is a low-cost multimodal reasoning model with 256K context, image and video input, function tools, and structured output.", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-02", "last_updated": "2026-03-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.04, "output": 0.07, "cache_read": 0.02 } }, "gemma-4-26b-a4b": { "id": "gemma-4-26b-a4b", "name": "Gemma 4 26B-A4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.05, "output": 0.29, "cache_read": 0.025 } }, "qwen3-5-9b": { "id": "qwen3-5-9b", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.09, "output": 0.13, "cache_read": 0.045 } }, "qwen3-7-max": { "id": "qwen3-7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 64000 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-06-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5 } }, "qwen3-6-max-preview": { "id": "qwen3-6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 393216 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 1.31, "output": 7.88, "tiers": [ { "input": 1.97, "output": 11.82, "tier": { "type": "context", "size": 128000 } } ] } }, "mimo-v2-5-pro": { "id": "mimo-v2-5-pro", "name": "MiMo V2.5 Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2.175, "output": 4.35, "cache_read": 0.018 } }, "qwen3-5-122b-a10b": { "id": "qwen3-5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1, "max": 80000 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.115, "output": 0.917, "tiers": [ { "input": 0.287, "output": 2.294, "tier": { "type": "context", "size": 128000 } } ] } }, "fugu-ultra": { "id": "fugu-ultra", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "family": "fugu", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-15", "last_updated": "2026-06-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 7.5, "output": 45, "cache_read": 1.5, "tiers": [ { "input": 15, "output": 67.5, "cache_read": 1.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 15, "output": 67.5, "cache_read": 1.5 } } }, "muse-spark-1-1": { "id": "muse-spark-1-1", "name": "Muse Spark 1.1", "description": "Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.", "family": "muse", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.25, "output": 4.25, "cache_read": 1 } } } }, "ovhcloud": { "id": "ovhcloud", "env": ["OVHCLOUD_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://oai.endpoints.kepler.ai.cloud.ovh.net/v1", "name": "OVHcloud AI Endpoints", "doc": "https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog//", "models": { "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder-30B-A3B-Instruct", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.07, "output": 0.26 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3-32B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-16", "last_updated": "2025-07-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.09, "output": 0.25 } }, "qwen3guard-gen-8b": { "id": "qwen3guard-gen-8b", "name": "Qwen3Guard-Gen-8B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 } }, "qwen3guard-gen-0.6b": { "id": "qwen3guard-gen-0.6b", "name": "Qwen3Guard-Gen-0.6B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 } }, "meta-llama-3_3-70b-instruct": { "id": "meta-llama-3_3-70b-instruct", "name": "Meta-Llama-3_3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-01", "last_updated": "2025-04-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.74, "output": 0.74 } }, "mistral-small-3.2-24b-instruct-2506": { "id": "mistral-small-3.2-24b-instruct-2506", "name": "Mistral-Small-3.2-24B-Instruct-2506", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-16", "last_updated": "2025-07-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.1, "output": 0.31 } }, "qwen2.5-vl-72b-instruct": { "id": "qwen2.5-vl-72b-instruct", "name": "Qwen2.5-VL-72B-Instruct", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-03-31", "last_updated": "2025-03-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 1.01, "output": 1.01 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.09, "output": 0.47 } }, "mistral-7b-instruct-v0.3": { "id": "mistral-7b-instruct-v0.3", "name": "Mistral-7B-Instruct-v0.3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-01", "last_updated": "2025-04-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.11, "output": 0.11 } }, "mistral-nemo-instruct-2407": { "id": "mistral-nemo-instruct-2407", "name": "Mistral-Nemo-Instruct-2407", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.14, "output": 0.14 } }, "qwen3.6-27b": { "id": "qwen3.6-27b", "name": "Qwen3.6-27B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.47, "output": 3.19 } }, "qwen3.5-9b": { "id": "qwen3.5-9b", "name": "Qwen3.5-9B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.12, "output": 0.18 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen3.5-397B-A17B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-18", "last_updated": "2026-05-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.71, "output": 4.25 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "gpt-oss-20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.18 } } } }, "friendli": { "id": "friendli", "env": ["FRIENDLI_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.friendli.ai/serverless/v1", "name": "Friendli", "doc": "https://friendli.ai/docs/guides/serverless_endpoints/introduction", "models": { "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.14, "output": 0.4 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-29", "last_updated": "2026-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2, "output": 0.8 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 0.25 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } } } }, "tencent-tokenhub": { "id": "tencent-tokenhub", "env": ["TENCENT_TOKENHUB_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://tokenhub.tencentmaas.com/v1", "name": "Tencent TokenHub", "doc": "https://cloud.tencent.com/document/product/1823/130050", "models": { "hy3": { "id": "hy3", "name": "Hy3", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "hy3-preview": { "id": "hy3-preview", "name": "Hy3 preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "wandb": { "id": "wandb", "env": ["WANDB_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.inference.wandb.ai/v1", "name": "Weights & Biases", "doc": "https://docs.wandb.ai/guides/integrations/inference/", "models": { "ibm-granite/granite-4.1-8b": { "id": "ibm-granite/granite-4.1-8b", "name": "Granite 4.1 8B", "description": "Granite 4.1 8B is a long-context instruct model capable of enhanced tool calling, instruction following, and chat capabilities.", "family": "granite", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.1, "cache_read": 0.05 } }, "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama 3.3 70B", "description": "Multilingual model excelling in conversational tasks, detailed instruction-following, and coding.", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.71, "output": 0.71, "cache_read": 0.71 } }, "meta-llama/Llama-3.1-70B-Instruct": { "id": "meta-llama/Llama-3.1-70B-Instruct", "name": "Llama 3.1 70B", "description": "Efficient conversational model optimized for responsive multilingual chatbot interactions.", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.8, "output": 0.8, "cache_read": 0.8 } }, "meta-llama/Llama-3.1-8B-Instruct": { "id": "meta-llama/Llama-3.1-8B-Instruct", "name": "Llama 3.1 8B", "description": "Efficient conversational model optimized for responsive multilingual chatbot interactions.", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.22, "output": 0.22, "cache_read": 0.22 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Kimi K2.6 is a multimodal Mixture-of-Experts language model featuring 32 billion activated parameters and a total of 1 trillion parameters.", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi K2.5", "description": "Kimi K2.5 is a multimodal Mixture-of-Experts language model featuring 32 billion activated parameters and a total of 1 trillion parameters.", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-02", "last_updated": "2026-02-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Kimi K2.7 Code is a 1T-parameter MoE model with 32B active parameters purpose-built for long-horizon agentic coding and software engineering.", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.94, "output": 4, "cache_read": 0.19 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B", "description": "Gemma 4 31B Dense is designed for advanced reasoning, agentic workflows, and longer context and is natively trained on 140+ languages.", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.12, "output": 0.35, "cache_read": 0.09 } }, "microsoft/Phi-4-mini-instruct": { "id": "microsoft/Phi-4-mini-instruct", "name": "Phi 4 Mini 3.8B", "description": "Compact, efficient model ideal for fast responses in resource-constrained environments.", "family": "phi", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-10", "release_date": "2025-02-01", "last_updated": "2025-02-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "status": "deprecated", "cost": { "input": 0.08, "output": 0.35, "cache_read": 0.08 } }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B A3B", "description": "Qwen3.6-35B-A3B is an MoE multimodal model with 262K context optimized for agentic coding workflows.", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-15", "last_updated": "2026-04-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.25 } }, "Qwen/Qwen3.6-27B": { "id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen3.6-27B is a 27B dense multimodal model with 262K context built for flagship-level agentic coding.", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3.6, "cache_read": 0.12 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen3 235B A22B Thinking-2507", "description": "High-performance Mixture-of-Experts model optimized for structured reasoning, math, and long-form generation.", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.1, "cache_read": 0.1 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen3 Coder 480B A35B", "description": "Mixture-of-Experts model optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning.", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1, "output": 1.5, "cache_read": 1 } }, "Qwen/Qwen3.5-27B": { "id": "Qwen/Qwen3.5-27B", "name": "Qwen3.5-27B", "description": "Qwen3.5-27B is a dense model from the Qwen3.5 family built for high performance across a large range of benchmarks.", "family": "qwen3.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.39, "output": 3.12, "cache_read": 0.08 } }, "Qwen/Qwen3-30B-A3B-Instruct-2507": { "id": "Qwen/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen3 30B A3B Instruct 2507", "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B MoE instruction-tuned model with enhanced reasoning, coding, and long-context understanding.", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.1 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B A22B-2507", "description": "Efficient multilingual, Mixture-of-Experts, instruction-tuned model, optimized for logical reasoning.", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.1, "output": 0.1, "cache_read": 0.1 } }, "Qwen/Qwen3.5-35B-A3B": { "id": "Qwen/Qwen3.5-35B-A3B", "name": "Qwen3.5-35B-A3B", "description": "Qwen3.5-35B-A3B is an open-weights multimodal MoE model built for efficient, high-throughput inference across chat, reasoning, and agentic tasks.", "family": "qwen3.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.25 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "gpt-oss-120b", "description": "Efficient Mixture-of-Experts model designed for high-reasoning, agentic and general-purpose use cases.", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.04, "output": 0.14, "cache_read": 0.04 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "gpt-oss-20b", "description": "Lower latency Mixture-of-Experts model trained on OpenAI's Harmony response format with reasoning capabilities.", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.03, "output": 0.13, "cache_read": 0.03 } }, "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", "name": "Nemotron 3 Ultra", "description": "Nemotron 3 Ultra is a powerful MoE model designed for long-running agents across coding, deep research, and enterprise automation.", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.75, "output": 2.75, "cache_read": 0.15 } }, "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8": { "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", "name": "Nemotron 3 Super", "description": "Nemotron 3 is a LatentMoE model designed to deliver strong agentic, reasoning, and conversational capabilities.", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2, "output": 0.8, "cache_read": 0.2 } }, "JetBrains/Mellum2-12B-A2.5B-Instruct": { "id": "JetBrains/Mellum2-12B-A2.5B-Instruct", "name": "Mellum2 12B A2.5B", "description": "Mellum2-12B-A2.5B-Instruct is a fast MoE model with 131K context built for coding, tool use, and low-latency AI workflows.", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.1, "cache_read": 0.05 } }, "OpenPipe/Qwen3-14B-Instruct": { "id": "OpenPipe/Qwen3-14B-Instruct", "name": "Qwen3 14B Instruct", "description": "An efficient multilingual, dense, instruction-tuned model, optimized by OpenPipe for building agents with finetuning.", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.05, "output": 0.22, "cache_read": 0.05 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM 5.2", "description": "GLM-5.2 is a Mixture-of-Experts language model featuring 40 billion activated parameters and a total of 744 billion parameters.", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-16", "last_updated": "2026-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.39, "output": 4.4, "cache_read": 0.26 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM 5.1", "description": "Powerful MoE model for long-horizon agentic engineering and advanced reasoning.", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "deepseek-ai/DeepSeek-V3.1": { "id": "deepseek-ai/DeepSeek-V3.1", "name": "DeepSeek V3.1", "description": "A large hybrid model that supports both thinking and non-thinking modes via prompt templates.", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 161000, "output": 161000 }, "cost": { "input": 0.55, "output": 1.65, "cache_read": 0.55 } }, "deepseek-ai/DeepSeek-V4-Flash": { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "DeepSeek V4-Flash is an MoE model with 1M context length great for coding, reasoning, and agentic workloads.", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 1048576 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.07 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "DeepSeek V4-Pro is a 1.6T-parameter MoE model with 49B active parameters excelling at advanced reasoning, coding, and complex agentic workloads.", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 1048576 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.14 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax M2.5", "description": "MoE model with a highly sparse architecture designed for high-throughput and low latency with strong coding capabilities.", "family": "minimax-m2.5", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.3 } } } }, "kuae-cloud-coding-plan": { "id": "kuae-cloud-coding-plan", "env": ["KUAE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://coding-plan-endpoint.kuaecloud.net/v1", "name": "KUAE Cloud Coding Plan", "doc": "https://docs.mthreads.com/kuaecloud/kuaecloud-doc-online/coding_plan/", "models": { "GLM-4.7": { "id": "GLM-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "gitlab": { "id": "gitlab", "env": ["GITLAB_TOKEN"], "npm": "gitlab-ai-provider", "name": "GitLab Duo", "doc": "https://docs.gitlab.com/user/duo_agent_platform/", "models": { "duo-chat-opus-4-5": { "id": "duo-chat-opus-4-5", "name": "Agentic Chat (Claude Opus 4.5)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2026-01-08", "last_updated": "2026-01-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-opus-4-8": { "id": "duo-chat-opus-4-8", "name": "Agentic Chat (Claude Opus 4.8)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-opus-4-7": { "id": "duo-chat-opus-4-7", "name": "Agentic Chat (Claude Opus 4.7)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-gpt-5-2-codex": { "id": "duo-chat-gpt-5-2-codex", "name": "Agentic Chat (GPT-5.2 Codex)", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-fable-5": { "id": "duo-chat-fable-5", "name": "Agentic Chat (Claude Fable 5)", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-gpt-5-5": { "id": "duo-chat-gpt-5-5", "name": "Agentic Chat (GPT-5.5)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-opus-4-6": { "id": "duo-chat-opus-4-6", "name": "Agentic Chat (Claude Opus 4.6)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-gpt-5-4": { "id": "duo-chat-gpt-5-4", "name": "Agentic Chat (GPT-5.4)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-gpt-5-codex": { "id": "duo-chat-gpt-5-codex", "name": "Agentic Chat (GPT-5 Codex)", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-gpt-5-4-nano": { "id": "duo-chat-gpt-5-4-nano", "name": "Agentic Chat (GPT-5.4 Nano)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-sonnet-4-6": { "id": "duo-chat-sonnet-4-6", "name": "Agentic Chat (Claude Sonnet 4.6)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-gpt-5-mini": { "id": "duo-chat-gpt-5-mini", "name": "Agentic Chat (GPT-5 Mini)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-sonnet-5": { "id": "duo-chat-sonnet-5", "name": "Agentic Chat (Claude Sonnet 5)", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-gpt-5-4-mini": { "id": "duo-chat-gpt-5-4-mini", "name": "Agentic Chat (GPT-5.4 Mini)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-gpt-5-3-codex": { "id": "duo-chat-gpt-5-3-codex", "name": "Agentic Chat (GPT-5.3 Codex)", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-haiku-4-5": { "id": "duo-chat-haiku-4-5", "name": "Agentic Chat (Claude Haiku 4.5)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2026-01-08", "last_updated": "2026-01-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-gpt-5-2": { "id": "duo-chat-gpt-5-2", "name": "Agentic Chat (GPT-5.2)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-01-23", "last_updated": "2026-01-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "duo-chat-sonnet-4-5": { "id": "duo-chat-sonnet-4-5", "name": "Agentic Chat (Claude Sonnet 4.5)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2026-01-08", "last_updated": "2026-01-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "duo-chat-gpt-5-1": { "id": "duo-chat-gpt-5-1", "name": "Agentic Chat (GPT-5.1)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0, "output": 0 } } } }, "kilo": { "id": "kilo", "env": ["KILO_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.kilo.ai/api/gateway", "name": "Kilo Gateway", "doc": "https://kilo.ai", "models": { "inclusionai/ling-2.6-1t": { "id": "inclusionai/ling-2.6-1t", "name": "inclusionAI: Ling-2.6-1T", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-04-23", "last_updated": "2026-05-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.06 } }, "inclusionai/ring-2.6-1t": { "id": "inclusionai/ring-2.6-1t", "name": "inclusionAI: Ring-2.6-1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-05-08", "last_updated": "2026-05-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.075, "output": 0.625, "cache_read": 0.015 } }, "inclusionai/ling-2.6-flash": { "id": "inclusionai/ling-2.6-flash", "name": "inclusionAI: Ling-2.6 Flash", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.08, "output": 0.24, "cache_read": 0.016 } }, "ibm-granite/granite-4.0-h-micro": { "id": "ibm-granite/granite-4.0-h-micro", "name": "IBM: Granite 4.0 Micro", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-20", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 32768 }, "cost": { "input": 0.017, "output": 0.11 } }, "ibm-granite/granite-4.1-8b": { "id": "ibm-granite/granite-4.1-8b", "name": "IBM: Granite 4.1 8B", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.1, "cache_read": 0.05 } }, "meta-llama/llama-3.1-8b-instruct": { "id": "meta-llama/llama-3.1-8b-instruct", "name": "Meta: Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.02, "output": 0.05 } }, "meta-llama/llama-3-70b-instruct": { "id": "meta-llama/llama-3-70b-instruct", "name": "Meta: Llama 3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8000 }, "cost": { "input": 0.51, "output": 0.74 } }, "meta-llama/llama-3.1-70b-instruct": { "id": "meta-llama/llama-3.1-70b-instruct", "name": "Meta: Llama 3.1 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-16", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.4, "output": 0.4 } }, "meta-llama/llama-3.2-1b-instruct": { "id": "meta-llama/llama-3.2-1b-instruct", "name": "Meta: Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-09-18", "last_updated": "2026-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 60000, "output": 12000 }, "cost": { "input": 0.027, "output": 0.2 } }, "meta-llama/llama-4-maverick": { "id": "meta-llama/llama-4-maverick", "name": "Meta: Llama 4 Maverick", "description": "Open multimodal Llama model for strong reasoning and fast responses", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-05", "last_updated": "2025-12-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "meta-llama/llama-3.2-11b-vision-instruct": { "id": "meta-llama/llama-3.2-11b-vision-instruct", "name": "Meta: Llama 3.2 11B Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.049, "output": 0.049 } }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", "name": "Meta: Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-08-01", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.1, "output": 0.32 } }, "meta-llama/llama-guard-3-8b": { "id": "meta-llama/llama-guard-3-8b", "name": "Llama Guard 3 8B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-18", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.02, "output": 0.06 } }, "meta-llama/llama-guard-4-12b": { "id": "meta-llama/llama-guard-4-12b", "name": "Meta: Llama Guard 4 12B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.18, "output": 0.18 } }, "meta-llama/llama-3-8b-instruct": { "id": "meta-llama/llama-3-8b-instruct", "name": "Meta: Llama 3 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-04-25", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 16384 }, "cost": { "input": 0.03, "output": 0.04 } }, "meta-llama/llama-4-scout": { "id": "meta-llama/llama-4-scout", "name": "Meta: Llama 4 Scout", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 327680, "output": 16384 }, "cost": { "input": 0.08, "output": 0.3 } }, "meta-llama/llama-3.2-3b-instruct": { "id": "meta-llama/llama-3.2-3b-instruct", "name": "Meta: Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-09-18", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 80000, "output": 16384 }, "cost": { "input": 0.051, "output": 0.34 } }, "~anthropic/claude-haiku-latest": { "id": "~anthropic/claude-haiku-latest", "name": "Anthropic: Claude Haiku Latest", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "~anthropic/claude-sonnet-latest": { "id": "~anthropic/claude-sonnet-latest", "name": "Anthropic: Claude Sonnet Latest", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "~anthropic/claude-opus-latest": { "id": "~anthropic/claude-opus-latest", "name": "Anthropic: Claude Opus Latest", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "kilo-auto/balanced": { "id": "kilo-auto/balanced", "name": "Kilo Auto Balanced", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 3 } }, "kilo-auto/small": { "id": "kilo-auto/small", "name": "Kilo Auto Small", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4 } }, "kilo-auto/free": { "id": "kilo-auto/free", "name": "Kilo Auto Free", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "kilo-auto/frontier": { "id": "kilo-auto/frontier", "name": "Kilo Auto Frontier", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "moonshotai/kimi-k2": { "id": "moonshotai/kimi-k2", "name": "MoonshotAI: Kimi K2 0711", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-11", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 26215 }, "cost": { "input": 0.55, "output": 2.2 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "MoonshotAI: Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-11-06", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65535 }, "cost": { "input": 0.47, "output": 2, "cache_read": 0.2 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "MoonshotAI: Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65535 }, "cost": { "input": 0.45, "output": 2.2 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "MoonshotAI: Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-05-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65535 }, "cost": { "input": 0.75, "output": 3.5, "cache_read": 0.375 } }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", "name": "MoonshotAI: Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.15 } }, "baidu/ernie-4.5-300b-a47b": { "id": "baidu/ernie-4.5-300b-a47b", "name": "Baidu: ERNIE 4.5 300B A47B ", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-06-30", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 123000, "output": 12000 }, "cost": { "input": 0.28, "output": 1.1 } }, "baidu/ernie-4.5-vl-28b-a3b": { "id": "baidu/ernie-4.5-vl-28b-a3b", "name": "Baidu: ERNIE 4.5 VL 28B A3B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 30000, "output": 8000 }, "cost": { "input": 0.14, "output": 0.56 } }, "baidu/ernie-4.5-vl-424b-a47b": { "id": "baidu/ernie-4.5-vl-424b-a47b", "name": "Baidu: ERNIE 4.5 VL 424B A47B ", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2025-06-30", "last_updated": "2026-01", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 123000, "output": 16000 }, "cost": { "input": 0.42, "output": 1.25 } }, "baidu/ernie-4.5-21b-a3b-thinking": { "id": "baidu/ernie-4.5-21b-a3b-thinking", "name": "Baidu: ERNIE 4.5 21B A3B Thinking", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.07, "output": 0.28 } }, "baidu/cobuddy:free": { "id": "baidu/cobuddy:free", "name": "Baidu: CoBuddy (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "release_date": "2026-05-06", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "baidu/ernie-4.5-21b-a3b": { "id": "baidu/ernie-4.5-21b-a3b", "name": "Baidu: ERNIE 4.5 21B A3B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 120000, "output": 8000 }, "cost": { "input": 0.07, "output": 0.28 } }, "baidu/qianfan-ocr-fast": { "id": "baidu/qianfan-ocr-fast", "name": "Baidu: Qianfan-OCR-Fast", "description": "OCR model for extracting structured text from documents and screenshots", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-05-16", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "output": 28672 }, "cost": { "input": 0.68, "output": 2.81 } }, "perceptron/perceptron-mk1": { "id": "perceptron/perceptron-mk1", "name": "Perceptron: Perceptron Mk1", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-05-12", "last_updated": "2026-05-16", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0.15, "output": 1.5 } }, "alfredpros/codellama-7b-instruct-solidity": { "id": "alfredpros/codellama-7b-instruct-solidity", "name": "AlfredPros: CodeLLaMa 7B Instruct Solidity", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-14", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 0.8, "output": 1.2 } }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Google: Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-05-07", "last_updated": "2026-05-16", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "reasoning": 1.5, "cache_read": 0.025, "cache_write": 0.08333 } }, "google/gemma-3n-e4b-it": { "id": "google/gemma-3n-e4b-it", "name": "Google: Gemma 3n 4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 6554 }, "cost": { "input": 0.02, "output": 0.04 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Google: Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "temperature": true, "release_date": "2025-03-20", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375 } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Google: Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": true, "release_date": "2025-07-17", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.3, "output": 2.5, "reasoning": 2.5, "cache_read": 0.03, "cache_write": 0.083333 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Google: Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-05-19", "last_updated": "2026-05-27", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "reasoning": 9, "cache_read": 0.15, "cache_write": 0.08333 } }, "google/gemini-2.0-flash-lite-001": { "id": "google/gemini-2.0-flash-lite-001", "name": "Google: Gemini 2.0 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-11", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 8192 }, "cost": { "input": 0.075, "output": 0.3 } }, "google/gemini-2.5-flash-lite-preview-09-2025": { "id": "google/gemini-2.5-flash-lite-preview-09-2025", "name": "Google: Gemini 2.5 Flash Lite Preview 09-2025", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2025-09-25", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "reasoning": 0.4, "cache_read": 0.01, "cache_write": 0.083333 } }, "google/gemini-2.0-flash-001": { "id": "google/gemini-2.0-flash-001", "name": "Google: Gemini 2.0 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-11", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 8192 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025, "cache_write": 0.083333 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Google: Gemma 4 31B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.14, "output": 0.4 } }, "google/lyria-3-clip-preview": { "id": "google/lyria-3-clip-preview", "name": "Google: Lyria 3 Clip Preview", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-30", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text"], "output": ["audio", "text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "google/gemini-3.1-pro-preview-customtools": { "id": "google/gemini-3.1-pro-preview-customtools", "name": "Google: Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "reasoning": 12 } }, "google/gemini-3-pro-image-preview": { "id": "google/gemini-3-pro-image-preview", "name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": false, "temperature": true, "release_date": "2025-11-20", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 65536, "output": 32768 }, "cost": { "input": 2, "output": 12, "reasoning": 12 } }, "google/gemini-2.5-flash-image": { "id": "google/gemini-2.5-flash-image", "name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-08", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.3, "output": 2.5 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Google: Gemini 2.5 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": true, "release_date": "2025-06-17", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.1, "output": 0.4, "reasoning": 0.4, "cache_read": 0.01, "cache_write": 0.083333 } }, "google/gemini-3.1-flash-image-preview": { "id": "google/gemini-3.1-flash-image-preview", "name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": false, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.5, "output": 3 } }, "google/gemini-2.5-pro-preview-05-06": { "id": "google/gemini-2.5-pro-preview-05-06", "name": "Google: Gemini 2.5 Pro Preview 05-06", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2025-05-06", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Google: Gemini 3.1 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-19", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "reasoning": 12 } }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Google: Gemma 4 26B A4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-04-03", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.12, "output": 0.4 } }, "google/gemini-2.5-pro-preview": { "id": "google/gemini-2.5-pro-preview", "name": "Google: Gemini 2.5 Pro Preview 06-05", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2025-06-05", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375 } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Google: Gemini 3 Flash Preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2025-12-17", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "reasoning": 3, "cache_read": 0.05, "cache_write": 0.083333 } }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", "name": "Google: Gemma 3 12B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-13", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.04, "output": 0.13, "cache_read": 0.015 } }, "google/gemma-3-4b-it": { "id": "google/gemma-3-4b-it", "name": "Google: Gemma 3 4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-13", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 19200 }, "cost": { "input": 0.04, "output": 0.08 } }, "google/gemma-3-27b-it": { "id": "google/gemma-3-27b-it", "name": "Google: Gemma 3 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-03-12", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 65536 }, "cost": { "input": 0.03, "output": 0.11, "cache_read": 0.02 } }, "google/lyria-3-pro-preview": { "id": "google/lyria-3-pro-preview", "name": "Google: Lyria 3 Pro Preview", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-30", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text"], "output": ["audio", "text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "google/gemma-2-27b-it": { "id": "google/gemma-2-27b-it", "name": "Google: Gemma 2 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-06-24", "last_updated": "2024-06-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0.65, "output": 0.65 } }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", "name": "Google: Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "reasoning": 1.5 } }, "liquid/lfm-2-24b-a2b": { "id": "liquid/lfm-2-24b-a2b", "name": "LiquidAI: LFM2-24B-A2B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.03, "output": 0.12 } }, "x-ai/grok-4.20": { "id": "x-ai/grok-4.20", "name": "xAI: Grok 4.20", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-31", "last_updated": "2026-04-11", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2 } }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", "name": "xAI: Grok 4.3", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 4096 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "x-ai/grok-4.20-multi-agent": { "id": "x-ai/grok-4.20-multi-agent", "name": "xAI: Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": false, "temperature": true, "release_date": "2026-03-31", "last_updated": "2026-04-11", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2 } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "xAI: Grok Build 0.1", "description": "Grok coding model for agentic engineering, edits, and codebase workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-05-20", "last_updated": "2026-05-27", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2 } }, "~google/gemini-pro-latest": { "id": "~google/gemini-pro-latest", "name": "Google: Gemini Pro Latest", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "cache_write": 0.375 } }, "~google/gemini-flash-latest": { "id": "~google/gemini-flash-latest", "name": "Google: Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.08333333333333334 } }, "microsoft/phi-4-mini-instruct": { "id": "microsoft/phi-4-mini-instruct", "name": "Microsoft: Phi 4 Mini Instruct", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-10-17", "last_updated": "2026-05-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.08, "output": 0.35, "cache_read": 0.08 } }, "microsoft/phi-4": { "id": "microsoft/phi-4", "name": "Microsoft: Phi 4", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.06, "output": 0.14 } }, "microsoft/wizardlm-2-8x22b": { "id": "microsoft/wizardlm-2-8x22b", "name": "WizardLM-2 8x22B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-24", "last_updated": "2024-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65535, "output": 8000 }, "cost": { "input": 0.62, "output": 0.62 } }, "poolside/laguna-xs.2:free": { "id": "poolside/laguna-xs.2:free", "name": "Poolside: Laguna XS.2 (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "poolside/laguna-m.1:free": { "id": "poolside/laguna-m.1:free", "name": "Poolside: Laguna M.1 (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "writer/palmyra-x5": { "id": "writer/palmyra-x5", "name": "Writer: Palmyra X5", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1040000, "output": 8192 }, "cost": { "input": 0.6, "output": 6 } }, "z-ai/glm-4.7": { "id": "z-ai/glm-4.7", "name": "Z.ai: GLM 4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2025-12-22", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 65535 }, "cost": { "input": 0.38, "output": 1.98, "cache_read": 0.2 } }, "z-ai/glm-4.5v": { "id": "z-ai/glm-4.5v", "name": "Z.ai: GLM 4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8, "cache_read": 0.11 } }, "z-ai/glm-4.5": { "id": "z-ai/glm-4.5", "name": "Z.ai: GLM 4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.175 } }, "z-ai/glm-5.1": { "id": "z-ai/glm-5.1", "name": "Z.ai: GLM 5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1.26, "output": 3.96 } }, "z-ai/glm-4.6": { "id": "z-ai/glm-4.6", "name": "Z.ai: GLM 4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2025-09-30", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 204800 }, "cost": { "input": 0.39, "output": 1.9, "cache_read": 0.175 } }, "z-ai/glm-4-32b": { "id": "z-ai/glm-4-32b", "name": "Z.ai: GLM 4 32B ", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-25", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.1 } }, "z-ai/glm-4.6v": { "id": "z-ai/glm-4.6v", "name": "Z.ai: GLM 4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2025-09-30", "last_updated": "2026-01-10", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.3, "output": 0.9 } }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "Z.ai: GLM 5V Turbo", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "z-ai/glm-4.5-air": { "id": "z-ai/glm-4.5-air", "name": "Z.ai: GLM 4.5 Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.13, "output": 0.85, "cache_read": 0.025 } }, "z-ai/glm-4.7-flash": { "id": "z-ai/glm-4.7-flash", "name": "Z.ai: GLM 4.7 Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 40551 }, "cost": { "input": 0.06, "output": 0.4, "cache_read": 0.01 } }, "z-ai/glm-5": { "id": "z-ai/glm-5", "name": "Z.ai: GLM 5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.72, "output": 2.3 } }, "z-ai/glm-5-turbo": { "id": "z-ai/glm-5-turbo", "name": "Z.ai: GLM 5 Turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "openai/gpt-4o-mini-2024-07-18": { "id": "openai/gpt-4o-mini-2024-07-18", "name": "OpenAI: GPT-4o-mini (2024-07-18)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-18", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "openai/gpt-oss-safeguard-20b": { "id": "openai/gpt-oss-safeguard-20b", "name": "OpenAI: gpt-oss-safeguard-20b", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-10-29", "last_updated": "2025-10-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.075, "output": 0.3, "cache_read": 0.037 } }, "openai/gpt-3.5-turbo-instruct": { "id": "openai/gpt-3.5-turbo-instruct", "name": "OpenAI: GPT-3.5 Turbo Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-03-01", "last_updated": "2023-09-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4095, "output": 4096 }, "cost": { "input": 1.5, "output": 2 } }, "openai/gpt-5.2-chat": { "id": "openai/gpt-5.2-chat", "name": "OpenAI: GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-12-11", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/o3": { "id": "openai/o3", "name": "OpenAI: o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-04-16", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/o4-mini-high": { "id": "openai/o4-mini-high", "name": "OpenAI: o4 Mini High", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "release_date": "2025-04-17", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4 } }, "openai/gpt-audio": { "id": "openai/gpt-audio", "name": "OpenAI: GPT Audio", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-20", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "text"], "output": ["audio", "text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "OpenAI: GPT-5.2 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2025-12-11", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "openai/gpt-4o-mini-search-preview": { "id": "openai/gpt-4o-mini-search-preview", "name": "OpenAI: GPT-4o-mini Search Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-01", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "OpenAI: GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-08-07", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5-chat": { "id": "openai/gpt-5-chat", "name": "OpenAI: GPT-5 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-08-07", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-3.5-turbo": { "id": "openai/gpt-3.5-turbo", "name": "OpenAI: GPT-3.5 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5 } }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", "name": "OpenAI: GPT-5 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-10-06", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 15, "output": 120 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "OpenAI: GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-05-13", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/gpt-4": { "id": "openai/gpt-4", "name": "OpenAI: GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-03-14", "last_updated": "2024-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 4096 }, "cost": { "input": 30, "output": 60 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "OpenAI: o4 Mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2025-04-16", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "openai/gpt-3.5-turbo-16k": { "id": "openai/gpt-3.5-turbo-16k", "name": "OpenAI: GPT-3.5 Turbo 16k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-08-28", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 3, "output": 4 } }, "openai/o3-pro": { "id": "openai/o3-pro", "name": "OpenAI: o3 Pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-04-16", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 20, "output": 80 } }, "openai/gpt-5.1-chat": { "id": "openai/gpt-5.1-chat", "name": "OpenAI: GPT-5.1 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-11-13", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-4o-2024-05-13": { "id": "openai/gpt-4o-2024-05-13", "name": "OpenAI: GPT-4o (2024-05-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-05-13", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 5, "output": 15 } }, "openai/gpt-4-0314": { "id": "openai/gpt-4-0314", "name": "OpenAI: GPT-4 (older v0314)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-05-28", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 4096 }, "cost": { "input": 30, "output": 60 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "OpenAI: GPT-5.4 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-03-17", "last_updated": "2026-04-11", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "openai/gpt-5.3-chat": { "id": "openai/gpt-5.3-chat", "name": "OpenAI: GPT-5.3 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2026-03-04", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14 } }, "openai/gpt-3.5-turbo-0613": { "id": "openai/gpt-3.5-turbo-0613", "name": "OpenAI: GPT-3.5 Turbo (older v0613)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-06-13", "last_updated": "2023-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4095, "output": 4096 }, "cost": { "input": 1, "output": 2 } }, "openai/gpt-5-image-mini": { "id": "openai/gpt-5-image-mini", "name": "OpenAI: GPT-5 Image Mini", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-16", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 2.5, "output": 2 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "OpenAI: GPT-5.1-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.1-codex-max": { "id": "openai/gpt-5.1-codex-max", "name": "OpenAI: GPT-5.1-Codex-Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-4o-2024-08-06": { "id": "openai/gpt-4o-2024-08-06", "name": "OpenAI: GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-08-06", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "OpenAI: o3 Mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-12-20", "last_updated": "2026-03-15", "modalities": { "input": ["pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "OpenAI: GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2025-12-11", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "OpenAI: GPT-5.3-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "release_date": "2026-02-25", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "openai/gpt-audio-mini": { "id": "openai/gpt-audio-mini", "name": "OpenAI: GPT Audio Mini", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-20", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "text"], "output": ["audio", "text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.6, "output": 2.4 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "OpenAI: GPT-5.1-Codex-Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 100000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/o4-mini-deep-research": { "id": "openai/o4-mini-deep-research", "name": "OpenAI: o4 Mini Deep Research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2024-06-26", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "OpenAI: GPT-4.1 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-14", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "OpenAI: gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.039, "output": 0.19 } }, "openai/gpt-4o-2024-11-20": { "id": "openai/gpt-4o-2024-11-20", "name": "OpenAI: GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-11-20", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/o1": { "id": "openai/o1", "name": "OpenAI: o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2024-12-05", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "openai/o1-pro": { "id": "openai/o1-pro", "name": "OpenAI: o1-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "temperature": false, "release_date": "2025-03-19", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 150, "output": 600 } }, "openai/gpt-chat-latest": { "id": "openai/gpt-chat-latest", "name": "OpenAI: GPT Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-05-05", "last_updated": "2026-05-07", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "openai/gpt-5-image": { "id": "openai/gpt-5-image", "name": "OpenAI: GPT-5 Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-14", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 10, "output": 10 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "OpenAI: GPT-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "release_date": "2026-03-06", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 2.5, "output": 15 } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "OpenAI: GPT-5.4 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-03-17", "last_updated": "2026-04-11", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "OpenAI: GPT-4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-14", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-4o-audio-preview": { "id": "openai/gpt-4o-audio-preview", "name": "OpenAI: GPT-4o Audio", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08-15", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "text"], "output": ["audio", "text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "openai/o3-deep-research": { "id": "openai/o3-deep-research", "name": "OpenAI: o3 Deep Research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "temperature": true, "release_date": "2024-06-26", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 10, "output": 40, "cache_read": 2.5 } }, "openai/gpt-4-turbo-preview": { "id": "openai/gpt-4-turbo-preview", "name": "OpenAI: GPT-4 Turbo Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-01-25", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "OpenAI: GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-08-07", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "OpenAI: GPT-4.1 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-14", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "openai/gpt-4-turbo": { "id": "openai/gpt-4-turbo", "name": "OpenAI: GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-09-13", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "OpenAI: GPT-5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-08-07", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "OpenAI: GPT-5.4 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "release_date": "2026-03-06", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "openai/o3-mini-high": { "id": "openai/o3-mini-high", "name": "OpenAI: o3 Mini High", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": false, "tool_call": true, "temperature": false, "release_date": "2025-01-31", "last_updated": "2026-03-15", "modalities": { "input": ["pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/gpt-5.4-image-2": { "id": "openai/gpt-5.4-image-2", "name": "OpenAI: GPT-5.4 Image 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-05-01", "modalities": { "input": ["image", "text", "pdf"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 8, "output": 15, "cache_read": 2 } }, "openai/gpt-4o-search-preview": { "id": "openai/gpt-4o-search-preview", "name": "OpenAI: GPT-4o Search Preview", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2025-03-13", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "OpenAI: GPT-5.5 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-24", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "OpenAI: GPT-4o-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-18", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "OpenAI: gpt-oss-20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.03, "output": 0.14 } }, "openai/gpt-4-1106-preview": { "id": "openai/gpt-4-1106-preview", "name": "OpenAI: GPT-4 Turbo (older v1106)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-11-06", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "OpenAI: GPT-5 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "OpenAI: GPT-5.2-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "OpenAI: GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "release_date": "2025-11-13", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "OpenAI: GPT-5.5", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-24", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "thedrummer/cydonia-24b-v4.1": { "id": "thedrummer/cydonia-24b-v4.1", "name": "TheDrummer: Cydonia 24B V4.1", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-27", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.3, "output": 0.5 } }, "thedrummer/skyfall-36b-v2": { "id": "thedrummer/skyfall-36b-v2", "name": "TheDrummer: Skyfall 36B V2", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-11", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.55, "output": 0.8 } }, "thedrummer/unslopnemo-12b": { "id": "thedrummer/unslopnemo-12b", "name": "TheDrummer: UnslopNemo 12B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-11-09", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.4, "output": 0.4 } }, "thedrummer/rocinante-12b": { "id": "thedrummer/rocinante-12b", "name": "TheDrummer: Rocinante 12B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-09-30", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.17, "output": 0.43 } }, "bytedance/ui-tars-1.5-7b": { "id": "bytedance/ui-tars-1.5-7b", "name": "ByteDance: UI-TARS 7B ", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-07-23", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 2048 }, "cost": { "input": 0.1, "output": 0.2 } }, "rekaai/reka-flash-3": { "id": "rekaai/reka-flash-3", "name": "Reka Flash 3", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-03-12", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.1, "output": 0.2 } }, "rekaai/reka-edge": { "id": "rekaai/reka-edge", "name": "Reka Edge", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-20", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.1, "output": 0.1 } }, "mistralai/mistral-large-2407": { "id": "mistralai/mistral-large-2407", "name": "Mistral Large 2407", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-11-19", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 2, "output": 6 } }, "mistralai/mistral-small-3.2-24b-instruct": { "id": "mistralai/mistral-small-3.2-24b-instruct", "name": "Mistral: Mistral Small 3.2 24B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.06, "output": 0.18, "cache_read": 0.03 } }, "mistralai/mistral-nemo": { "id": "mistralai/mistral-nemo", "name": "Mistral: Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-01", "last_updated": "2024-07-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.02, "output": 0.04 } }, "mistralai/mistral-medium-3-5": { "id": "mistralai/mistral-medium-3-5", "name": "Mistral: Mistral Medium 3.5", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-05-07", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.5, "output": 7.5 } }, "mistralai/ministral-8b-2512": { "id": "mistralai/ministral-8b-2512", "name": "Mistral: Ministral 3 8B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.15, "output": 0.15 } }, "mistralai/devstral-small": { "id": "mistralai/devstral-small", "name": "Mistral: Devstral Small 1.1", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-05-07", "last_updated": "2025-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.1, "output": 0.3 } }, "mistralai/mistral-small-3.1-24b-instruct": { "id": "mistralai/mistral-small-3.1-24b-instruct", "name": "Mistral: Mistral Small 3.1 24B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-17", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 131072 }, "cost": { "input": 0.35, "output": 0.56, "cache_read": 0.015 } }, "mistralai/mistral-saba": { "id": "mistralai/mistral-saba", "name": "Mistral: Saba", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-02-17", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.2, "output": 0.6 } }, "mistralai/mistral-large": { "id": "mistralai/mistral-large", "name": "Mistral Large", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-24", "last_updated": "2025-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 25600 }, "cost": { "input": 2, "output": 6 } }, "mistralai/mistral-medium-3.1": { "id": "mistralai/mistral-medium-3.1", "name": "Mistral: Mistral Medium 3.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.4, "output": 2 } }, "mistralai/pixtral-large-2411": { "id": "mistralai/pixtral-large-2411", "name": "Mistral: Pixtral Large 2411", "description": "Mistral vision-language model for image understanding and multimodal chat", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-11-19", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 2, "output": 6 } }, "mistralai/devstral-medium": { "id": "mistralai/devstral-medium", "name": "Mistral: Devstral Medium", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.4, "output": 2 } }, "mistralai/mistral-small-24b-instruct-2501": { "id": "mistralai/mistral-small-24b-instruct-2501", "name": "Mistral: Mistral Small 3", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-29", "last_updated": "2026-01-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.05, "output": 0.08 } }, "mistralai/ministral-3b-2512": { "id": "mistralai/ministral-3b-2512", "name": "Mistral: Ministral 3 3B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.1, "output": 0.1 } }, "mistralai/mistral-small-2603": { "id": "mistralai/mistral-small-2603", "name": "Mistral: Mistral Small 4", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015 } }, "mistralai/mistral-large-2411": { "id": "mistralai/mistral-large-2411", "name": "Mistral Large 2411", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-24", "last_updated": "2024-11-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 2, "output": 6 } }, "mistralai/mistral-7b-instruct-v0.1": { "id": "mistralai/mistral-7b-instruct-v0.1", "name": "Mistral: Mistral 7B Instruct v0.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2824, "output": 565 }, "cost": { "input": 0.11, "output": 0.19 } }, "mistralai/ministral-14b-2512": { "id": "mistralai/ministral-14b-2512", "name": "Mistral: Ministral 3 14B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 52429 }, "cost": { "input": 0.2, "output": 0.2 } }, "mistralai/devstral-2512": { "id": "mistralai/devstral-2512", "name": "Mistral: Devstral 2 2512", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-12", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.025 } }, "mistralai/mixtral-8x22b-instruct": { "id": "mistralai/mixtral-8x22b-instruct", "name": "Mistral: Mixtral 8x22B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-04-17", "last_updated": "2024-04-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 13108 }, "cost": { "input": 2, "output": 6 } }, "mistralai/mistral-medium-3": { "id": "mistralai/mistral-medium-3", "name": "Mistral: Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.4, "output": 2 } }, "mistralai/voxtral-small-24b-2507": { "id": "mistralai/voxtral-small-24b-2507", "name": "Mistral: Voxtral Small 24B 2507", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 6400 }, "cost": { "input": 0.1, "output": 0.3 } }, "mistralai/mistral-large-2512": { "id": "mistralai/mistral-large-2512", "name": "Mistral: Mistral Large 3 2512", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-11-01", "last_updated": "2025-12-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 52429 }, "cost": { "input": 0.5, "output": 1.5 } }, "mistralai/codestral-2508": { "id": "mistralai/codestral-2508", "name": "Mistral: Codestral 2508", "description": "Mistral coding model for code completion, generation, and developer workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08-01", "last_updated": "2025-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 51200 }, "cost": { "input": 0.3, "output": 0.9 } }, "morph/morph-v3-fast": { "id": "morph/morph-v3-fast", "name": "Morph: Morph V3 Fast", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-15", "last_updated": "2024-08-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 81920, "output": 38000 }, "cost": { "input": 0.8, "output": 1.2 } }, "morph/morph-v3-large": { "id": "morph/morph-v3-large", "name": "Morph: Morph V3 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-15", "last_updated": "2024-08-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.9, "output": 1.9 } }, "bytedance-seed/seed-1.6-flash": { "id": "bytedance-seed/seed-1.6-flash", "name": "ByteDance Seed: Seed 1.6 Flash", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.075, "output": 0.3 } }, "bytedance-seed/seed-1.6": { "id": "bytedance-seed/seed-1.6", "name": "ByteDance Seed: Seed 1.6", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.25, "output": 2 } }, "bytedance-seed/seed-2.0-mini": { "id": "bytedance-seed/seed-2.0-mini", "name": "ByteDance Seed: Seed-2.0-Mini", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-27", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.1, "output": 0.4 } }, "bytedance-seed/seed-2.0-lite": { "id": "bytedance-seed/seed-2.0-lite", "name": "ByteDance Seed: Seed-2.0-Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.25, "output": 2 } }, "anthracite-org/magnum-v4-72b": { "id": "anthracite-org/magnum-v4-72b", "name": "Magnum v4 72B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-22", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 2048 }, "cost": { "input": 3, "output": 5 } }, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free": { "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", "name": "NVIDIA: Nemotron 3 Nano Omni (free)", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-05-01", "modalities": { "input": ["text", "audio", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-nano-30b-a3b": { "id": "nvidia/nemotron-3-nano-30b-a3b", "name": "NVIDIA: Nemotron 3 Nano 30B A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2024-12", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 52429 }, "cost": { "input": 0.05, "output": 0.2 } }, "nvidia/llama-3.3-nemotron-super-49b-v1.5": { "id": "nvidia/llama-3.3-nemotron-super-49b-v1.5", "name": "NVIDIA: Llama 3.3 Nemotron Super 49B V1.5", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-03-16", "last_updated": "2025-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.1, "output": 0.4 } }, "nvidia/nemotron-nano-9b-v2": { "id": "nvidia/nemotron-nano-9b-v2", "name": "NVIDIA: Nemotron Nano 9B V2", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-18", "last_updated": "2025-08-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 0.04, "output": 0.16 } }, "nvidia/nemotron-3-super-120b-a12b:free": { "id": "nvidia/nemotron-3-super-120b-a12b:free", "name": "NVIDIA: Nemotron 3 Super (free)", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-12", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "NVIDIA: Nemotron 3 Super", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.1, "output": 0.5, "cache_read": 0.1 } }, "xiaomi/mimo-v2.5": { "id": "xiaomi/mimo-v2.5", "name": "Xiaomi: MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.08, "tiers": [ { "input": 0.8, "output": 4, "cache_read": 0.16, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.8, "output": 4, "cache_read": 0.16 } } }, "xiaomi/mimo-v2-omni": { "id": "xiaomi/mimo-v2-omni", "name": "Xiaomi: MiMo-V2-Omni", "description": "MiMo omni model for text, image, video, audio, and agents", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.08 } }, "xiaomi/mimo-v2-flash": { "id": "xiaomi/mimo-v2-flash", "name": "Xiaomi: MiMo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12-01", "release_date": "2025-12-16", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.09, "output": 0.29, "cache_read": 0.045 } }, "xiaomi/mimo-v2-pro": { "id": "xiaomi/mimo-v2-pro", "name": "Xiaomi: MiMo-V2-Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "xiaomi/mimo-v2.5-pro": { "id": "xiaomi/mimo-v2.5-pro", "name": "Xiaomi: MiMo V2.5 Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "inception/mercury-2": { "id": "inception/mercury-2", "name": "Inception: Mercury 2", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 50000 }, "cost": { "input": 0.25, "output": 0.75, "cache_read": 0.025 } }, "anthropic/claude-3.5-haiku": { "id": "anthropic/claude-3.5-haiku", "name": "Anthropic: Claude 3.5 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "anthropic/claude-sonnet-4.5": { "id": "anthropic/claude-sonnet-4.5", "name": "Anthropic: Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "release_date": "2025-09-29", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Anthropic: Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "release_date": "2025-05-22", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4.6-fast": { "id": "anthropic/claude-opus-4.6-fast", "name": "Anthropic: Claude Opus 4.6 (Fast)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-04-07", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 } }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Anthropic: Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-opus-4.7-fast": { "id": "anthropic/claude-opus-4.7-fast", "name": "Anthropic: Claude Opus 4.7 (Fast)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "release_date": "2026-05-12", "last_updated": "2026-05-16", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Anthropic: Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-opus-4.1": { "id": "anthropic/claude-opus-4.1", "name": "Anthropic: Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.5": { "id": "anthropic/claude-opus-4.5", "name": "Anthropic: Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "release_date": "2025-11-24", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Anthropic: Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15 } }, "anthropic/claude-3-haiku": { "id": "anthropic/claude-3-haiku", "name": "Anthropic: Claude 3 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-03-07", "last_updated": "2024-03-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.03, "cache_write": 0.3 } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Anthropic: Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "release_date": "2025-05-22", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Anthropic: Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "tencent/hunyuan-a13b-instruct": { "id": "tencent/hunyuan-a13b-instruct", "name": "Tencent: Hunyuan A13B Instruct", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-06-30", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.14, "output": 0.57 } }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Tencent: Hy3 Preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-05-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.066, "output": 0.26, "cache_read": 0.029 } }, "deepcogito/cogito-v2.1-671b": { "id": "deepcogito/cogito-v2.1-671b", "name": "Deep Cogito: Cogito v2.1 671B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-11-14", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 1.25, "output": 1.25 } }, "cohere/command-a": { "id": "cohere/command-a", "name": "Cohere: Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 2.5, "output": 10 } }, "cohere/command-r-08-2024": { "id": "cohere/command-r-08-2024", "name": "Cohere: Command R (08-2024)", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.15, "output": 0.6 } }, "cohere/command-r7b-12-2024": { "id": "cohere/command-r7b-12-2024", "name": "Cohere: Command R7B (12-2024)", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-02", "last_updated": "2024-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.0375, "output": 0.15 } }, "cohere/command-r-plus-08-2024": { "id": "cohere/command-r-plus-08-2024", "name": "Cohere: Command R+ (08-2024)", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 2.5, "output": 10 } }, "gryphe/mythomax-l2-13b": { "id": "gryphe/mythomax-l2-13b", "name": "MythoMax 13B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-25", "last_updated": "2024-04-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 0.06, "output": 0.06 } }, "stepfun/step-3.5-flash": { "id": "stepfun/step-3.5-flash", "name": "StepFun: Step 3.5 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2026-01-29", "last_updated": "2026-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "prime-intellect/intellect-3": { "id": "prime-intellect/intellect-3", "name": "Prime Intellect: INTELLECT-3", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-11-26", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.2, "output": 1.1 } }, "nex-agi/deepseek-v3.1-nex-n1": { "id": "nex-agi/deepseek-v3.1-nex-n1", "name": "Nex AGI: DeepSeek V3.1 Nex N1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 163840 }, "cost": { "input": 0.27, "output": 1 } }, "undi95/remm-slerp-l2-13b": { "id": "undi95/remm-slerp-l2-13b", "name": "ReMM SLERP 13B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-07-22", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 6144, "output": 4096 }, "cost": { "input": 0.45, "output": 0.65 } }, "~openai/gpt-mini-latest": { "id": "~openai/gpt-mini-latest", "name": "OpenAI: GPT Mini Latest", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "~openai/gpt-latest": { "id": "~openai/gpt-latest", "name": "OpenAI: GPT Latest", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "~moonshotai/kimi-latest": { "id": "~moonshotai/kimi-latest", "name": "MoonshotAI: Kimi Latest", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262142, "output": 262142 }, "cost": { "input": 0.74, "output": 3.49, "cache_read": 0.14 } }, "relace/relace-search": { "id": "relace/relace-search", "name": "Relace: Relace Search", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-09", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 1, "output": 3 } }, "relace/relace-apply-3": { "id": "relace/relace-apply-3", "name": "Relace: Relace Apply 3", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2025-09-26", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.85, "output": 1.25 } }, "ai21/jamba-large-1.7": { "id": "ai21/jamba-large-1.7", "name": "AI21: Jamba Large 1.7", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-08-09", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 4096 }, "cost": { "input": 2, "output": 8 } }, "arcee-ai/coder-large": { "id": "arcee-ai/coder-large", "name": "Arcee AI: Coder Large", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-06", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.5, "output": 0.8 } }, "arcee-ai/virtuoso-large": { "id": "arcee-ai/virtuoso-large", "name": "Arcee AI: Virtuoso Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-05-06", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 64000 }, "cost": { "input": 0.75, "output": 1.2 } }, "arcee-ai/spotlight": { "id": "arcee-ai/spotlight", "name": "Arcee AI: Spotlight", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-06", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65537 }, "cost": { "input": 0.18, "output": 0.18 } }, "arcee-ai/trinity-large-thinking": { "id": "arcee-ai/trinity-large-thinking", "name": "Arcee AI: Trinity Large Thinking", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.22, "output": 0.85 } }, "arcee-ai/maestro-reasoning": { "id": "arcee-ai/maestro-reasoning", "name": "Arcee AI: Maestro Reasoning", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-05-06", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32000 }, "cost": { "input": 0.9, "output": 3.3 } }, "arcee-ai/trinity-mini": { "id": "arcee-ai/trinity-mini", "name": "Arcee AI: Trinity Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12", "last_updated": "2026-01-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.045, "output": 0.15 } }, "mancer/weaver": { "id": "mancer/weaver", "name": "Mancer: Weaver (alpha)", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-08-02", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "output": 2000 }, "cost": { "input": 0.75, "output": 1 } }, "perplexity/sonar-reasoning-pro": { "id": "perplexity/sonar-reasoning-pro", "name": "Perplexity: Sonar Reasoning Pro", "description": "Web-grounded reasoning model for multi-step research and cited answers", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 25600 }, "cost": { "input": 2, "output": 8 } }, "perplexity/sonar": { "id": "perplexity/sonar", "name": "Perplexity: Sonar", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127072, "output": 25415 }, "cost": { "input": 1, "output": 1 } }, "perplexity/sonar-pro": { "id": "perplexity/sonar-pro", "name": "Perplexity: Sonar Pro", "description": "Advanced Sonar search model for deeper research and cited synthesis", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8000 }, "cost": { "input": 3, "output": 15 } }, "perplexity/sonar-pro-search": { "id": "perplexity/sonar-pro-search", "name": "Perplexity: Sonar Pro Search", "description": "Advanced Sonar search model for deeper research and cited synthesis", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-10-31", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8000 }, "cost": { "input": 3, "output": 15 } }, "perplexity/sonar-deep-research": { "id": "perplexity/sonar-deep-research", "name": "Perplexity: Sonar Deep Research", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 25600 }, "cost": { "input": 2, "output": 8 } }, "switchpoint/router": { "id": "switchpoint/router", "name": "Switchpoint Router", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-07-12", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.85, "output": 3.4 } }, "openrouter/bodybuilder": { "id": "openrouter/bodybuilder", "name": "Body Builder (beta)", "description": "Preview model for early access evaluation, prototyping, and compatibility testing", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "status": "beta", "cost": { "input": 0, "output": 0 } }, "openrouter/free": { "id": "openrouter/free", "name": "Free Models Router", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "openrouter/owl-alpha": { "id": "openrouter/owl-alpha", "name": "Owl Alpha", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "output": 262144 }, "status": "alpha", "cost": { "input": 0, "output": 0 } }, "openrouter/pareto-code": { "id": "openrouter/pareto-code", "name": "Pareto Code Router", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "openrouter/auto": { "id": "openrouter/auto", "name": "Auto Router", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["audio", "image", "pdf", "text", "video"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3.5-plus-20260420": { "id": "qwen/qwen3.5-plus-20260420", "name": "Qwen: Qwen3.5 Plus 2026-04-20", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.4, "output": 2.4 } }, "qwen/qwen3-vl-235b-a22b-thinking": { "id": "qwen/qwen3-vl-235b-a22b-thinking", "name": "Qwen: Qwen3 VL 235B A22B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-09-24", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.26, "output": 2.6 } }, "qwen/qwen3-vl-30b-a3b-thinking": { "id": "qwen/qwen3-vl-30b-a3b-thinking", "name": "Qwen: Qwen3 VL 30B A3B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-11", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.13, "output": 1.56 } }, "qwen/qwen3-coder-plus": { "id": "qwen/qwen3-coder-plus", "name": "Qwen: Qwen3 Coder Plus", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-01", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.65, "output": 3.25, "cache_read": 0.2 } }, "qwen/qwen-plus": { "id": "qwen/qwen-plus", "name": "Qwen: Qwen-Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-01-25", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.4, "output": 1.2, "cache_read": 0.08 } }, "qwen/qwen3-coder-30b-a3b-instruct": { "id": "qwen/qwen3-coder-30b-a3b-instruct", "name": "Qwen: Qwen3 Coder 30B A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 160000, "output": 32768 }, "cost": { "input": 0.07, "output": 0.27 } }, "qwen/qwen3-32b": { "id": "qwen/qwen3-32b", "name": "Qwen: Qwen3 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 40960 }, "cost": { "input": 0.08, "output": 0.24, "cache_read": 0.04 } }, "qwen/qwen3-next-80b-a3b-instruct": { "id": "qwen/qwen3-next-80b-a3b-instruct", "name": "Qwen: Qwen3 Next 80B A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-11", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 52429 }, "cost": { "input": 0.09, "output": 1.1 } }, "qwen/qwen3-vl-8b-instruct": { "id": "qwen/qwen3-vl-8b-instruct", "name": "Qwen: Qwen3 VL 8B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-10-15", "last_updated": "2025-11-25", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.08, "output": 0.5 } }, "qwen/qwen3.6-35b-a3b": { "id": "qwen/qwen3.6-35b-a3b", "name": "Qwen: Qwen3.6 35B A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.1612, "output": 0.96525, "cache_read": 0.1612 } }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen: Qwen3.7 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 262144 } ], "tool_call": true, "temperature": true, "release_date": "2025-08-26", "last_updated": "2026-05-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.625, "output": 4.875, "cache_read": 0.1625, "cache_write": 2.03125 } }, "qwen/qwen3-max": { "id": "qwen/qwen3-max", "name": "Qwen: Qwen3 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-05", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 1.2, "output": 6, "cache_read": 0.24 } }, "qwen/qwen3-8b": { "id": "qwen/qwen3-8b", "name": "Qwen: Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-04", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 8192 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.05 } }, "qwen/qwen-plus-2025-07-28": { "id": "qwen/qwen-plus-2025-07-28", "name": "Qwen: Qwen Plus 0728", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-09", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.26, "output": 0.78 } }, "qwen/qwen3.5-flash-02-23": { "id": "qwen/qwen3.5-flash-02-23", "name": "Qwen: Qwen3.5-Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4 } }, "qwen/qwen3-30b-a3b-instruct-2507": { "id": "qwen/qwen3-30b-a3b-instruct-2507", "name": "Qwen: Qwen3 30B A3B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-29", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.09, "output": 0.3, "cache_read": 0.04 } }, "qwen/qwen-2.5-coder-32b-instruct": { "id": "qwen/qwen-2.5-coder-32b-instruct", "name": "Qwen2.5 Coder 32B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-11-11", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0.2, "output": 0.2, "cache_read": 0.015 } }, "qwen/qwen3-next-80b-a3b-thinking": { "id": "qwen/qwen3-next-80b-a3b-thinking", "name": "Qwen: Qwen3 Next 80B A3B Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-09-11", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.0975, "output": 0.78 } }, "qwen/qwen3-235b-a22b-thinking-2507": { "id": "qwen/qwen3-235b-a22b-thinking-2507", "name": "Qwen: Qwen3 235B A22B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-07-25", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.11, "output": 0.6 } }, "qwen/qwen3-vl-32b-instruct": { "id": "qwen/qwen3-vl-32b-instruct", "name": "Qwen: Qwen3 VL 32B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-10-21", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.104, "output": 0.416 } }, "qwen/qwen3-coder": { "id": "qwen/qwen3-coder", "name": "Qwen: Qwen3 Coder 480B A35B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 52429 }, "cost": { "input": 0.22, "output": 1, "cache_read": 0.022 } }, "qwen/qwen3.6-flash": { "id": "qwen/qwen3.6-flash", "name": "Qwen: Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_write": 0.3125 } }, "qwen/qwen3.5-plus-02-15": { "id": "qwen/qwen3.5-plus-02-15", "name": "Qwen: Qwen3.5 Plus 2026-02-15", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.26, "output": 1.56 } }, "qwen/qwen-2.5-7b-instruct": { "id": "qwen/qwen-2.5-7b-instruct", "name": "Qwen: Qwen2.5 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-09", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 6554 }, "cost": { "input": 0.04, "output": 0.1 } }, "qwen/qwen3-vl-8b-thinking": { "id": "qwen/qwen3-vl-8b-thinking", "name": "Qwen: Qwen3 VL 8B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-15", "last_updated": "2025-11-25", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.117, "output": 1.365 } }, "qwen/qwen3-max-thinking": { "id": "qwen/qwen3-max-thinking", "name": "Qwen: Qwen3 Max Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2026-01-23", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.78, "output": 3.9 } }, "qwen/qwen3-30b-a3b-thinking-2507": { "id": "qwen/qwen3-30b-a3b-thinking-2507", "name": "Qwen: Qwen3 30B A3B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 6554 }, "cost": { "input": 0.051, "output": 0.34 } }, "qwen/qwen2.5-vl-72b-instruct": { "id": "qwen/qwen2.5-vl-72b-instruct", "name": "Qwen: Qwen2.5 VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-02-01", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.8, "output": 0.8, "cache_read": 0.075 } }, "qwen/qwen3.5-27b": { "id": "qwen/qwen3.5-27b", "name": "Qwen: Qwen3.5-27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.195, "output": 1.56 } }, "qwen/qwen3-235b-a22b": { "id": "qwen/qwen3-235b-a22b", "name": "Qwen: Qwen3 235B A22B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 38912 } ], "tool_call": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.455, "output": 1.82, "cache_read": 0.15 } }, "qwen/qwen-2.5-72b-instruct": { "id": "qwen/qwen-2.5-72b-instruct", "name": "Qwen2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-09", "last_updated": "2026-01-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.12, "output": 0.39 } }, "qwen/qwen-plus-2025-07-28:thinking": { "id": "qwen/qwen-plus-2025-07-28:thinking", "name": "Qwen: Qwen Plus 0728 (thinking)", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2025-09-09", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.26, "output": 0.78 } }, "qwen/qwen3-coder-next": { "id": "qwen/qwen3-coder-next", "name": "Qwen: Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-02-02", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.12, "output": 0.75, "cache_read": 0.035 } }, "qwen/qwen3.6-27b": { "id": "qwen/qwen3.6-27b", "name": "Qwen: Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.325, "output": 3.25 } }, "qwen/qwen3.5-35b-a3b": { "id": "qwen/qwen3.5-35b-a3b", "name": "Qwen: Qwen3.5-35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.1625, "output": 1.3 } }, "qwen/qwen3.5-9b": { "id": "qwen/qwen3.5-9b", "name": "Qwen: Qwen3.5-9B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.05, "output": 0.15 } }, "qwen/qwen3.5-397b-a17b": { "id": "qwen/qwen3.5-397b-a17b", "name": "Qwen: Qwen3.5 397B A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.39, "output": 2.34 } }, "qwen/qwen3-vl-30b-a3b-instruct": { "id": "qwen/qwen3-vl-30b-a3b-instruct", "name": "Qwen: Qwen3 VL 30B A3B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-10-05", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.13, "output": 0.52 } }, "qwen/qwen3-235b-a22b-2507": { "id": "qwen/qwen3-235b-a22b-2507", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 52429 }, "cost": { "input": 0.071, "output": 0.1 } }, "qwen/qwen3-coder-flash": { "id": "qwen/qwen3-coder-flash", "name": "Qwen: Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-23", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.195, "output": 0.975, "cache_read": 0.06 } }, "qwen/qwen3-14b": { "id": "qwen/qwen3-14b", "name": "Qwen: Qwen3 14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-04", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 40960 }, "cost": { "input": 0.06, "output": 0.24, "cache_read": 0.025 } }, "qwen/qwen3-vl-235b-a22b-instruct": { "id": "qwen/qwen3-vl-235b-a22b-instruct", "name": "Qwen: Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-09-23", "last_updated": "2026-01-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 52429 }, "cost": { "input": 0.2, "output": 0.88, "cache_read": 0.11 } }, "qwen/qwen3-30b-a3b": { "id": "qwen/qwen3-30b-a3b", "name": "Qwen: Qwen3 30B A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-04", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 40960 }, "cost": { "input": 0.08, "output": 0.28, "cache_read": 0.03 } }, "qwen/qwen3.6-max-preview": { "id": "qwen/qwen3.6-max-preview", "name": "Qwen: Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 131072 } ], "tool_call": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.04, "output": 6.24, "cache_write": 1.3 } }, "qwen/qwen3.5-122b-a10b": { "id": "qwen/qwen3.5-122b-a10b", "name": "Qwen: Qwen3.5-122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.26, "output": 2.08 } }, "qwen/qwen3.6-plus": { "id": "qwen/qwen3.6-plus", "name": "Qwen: Qwen3.6 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "temperature": true, "release_date": "2025-08-26", "last_updated": "2026-04-11", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.325, "output": 1.95, "cache_read": 0.0325, "cache_write": 0.40625 } }, "amazon/nova-lite-v1": { "id": "amazon/nova-lite-v1", "name": "Amazon: Nova Lite 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-06", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 5120 }, "cost": { "input": 0.06, "output": 0.24 } }, "amazon/nova-premier-v1": { "id": "amazon/nova-premier-v1", "name": "Amazon: Nova Premier 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-11-01", "last_updated": "2026-03-15", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 2.5, "output": 12.5 } }, "amazon/nova-pro-v1": { "id": "amazon/nova-pro-v1", "name": "Amazon: Nova Pro 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 5120 }, "cost": { "input": 0.8, "output": 3.2 } }, "amazon/nova-micro-v1": { "id": "amazon/nova-micro-v1", "name": "Amazon: Nova Micro 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-06", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 5120 }, "cost": { "input": 0.035, "output": 0.14 } }, "amazon/nova-2-lite-v1": { "id": "amazon/nova-2-lite-v1", "name": "Amazon: Nova 2 Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2026-03-15", "modalities": { "input": ["image", "pdf", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65535 }, "cost": { "input": 0.3, "output": 2.5 } }, "aion-labs/aion-rp-llama-3.1-8b": { "id": "aion-labs/aion-rp-llama-3.1-8b", "name": "AionLabs: Aion-RP 1.0 (8B)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-02-05", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.8, "output": 1.6 } }, "aion-labs/aion-1.0-mini": { "id": "aion-labs/aion-1.0-mini", "name": "AionLabs: Aion-1.0-Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-02-05", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.7, "output": 1.4 } }, "aion-labs/aion-2.0": { "id": "aion-labs/aion-2.0", "name": "AionLabs: Aion-2.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-02-24", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.8, "output": 1.6 } }, "aion-labs/aion-1.0": { "id": "aion-labs/aion-1.0", "name": "AionLabs: Aion-1.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-02-05", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 4, "output": 8 } }, "inflection/inflection-3-pi": { "id": "inflection/inflection-3-pi", "name": "Inflection: Inflection 3 Pi", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-11", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "output": 1024 }, "cost": { "input": 2.5, "output": 10 } }, "inflection/inflection-3-productivity": { "id": "inflection/inflection-3-productivity", "name": "Inflection: Inflection 3 Productivity", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-11", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "output": 1024 }, "cost": { "input": 2.5, "output": 10 } }, "sao10k/l3.1-euryale-70b": { "id": "sao10k/l3.1-euryale-70b", "name": "Sao10K: Llama 3.1 Euryale 70B v2.2", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-08-28", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.85, "output": 0.85 } }, "sao10k/l3.3-euryale-70b": { "id": "sao10k/l3.3-euryale-70b", "name": "Sao10K: Llama 3.3 Euryale 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-12-18", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.65, "output": 0.75 } }, "sao10k/l3-lunaris-8b": { "id": "sao10k/l3-lunaris-8b", "name": "Sao10K: Llama 3 8B Lunaris", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-13", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.04, "output": 0.05 } }, "sao10k/l3-euryale-70b": { "id": "sao10k/l3-euryale-70b", "name": "Sao10k: Llama 3 Euryale 70B v2.1", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-06-18", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 1.48, "output": 1.48 } }, "sao10k/l3.1-70b-hanami-x1": { "id": "sao10k/l3.1-70b-hanami-x1", "name": "Sao10K: Llama 3.1 70B Hanami x1", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-01-08", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 16000 }, "cost": { "input": 3, "output": 3 } }, "upstage/solar-pro-3": { "id": "upstage/solar-pro-3", "name": "Upstage: Solar Pro 3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6 } }, "allenai/olmo-3-32b-think": { "id": "allenai/olmo-3-32b-think", "name": "AllenAI: Olmo 3 32B Think", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-11-22", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.15, "output": 0.5 } }, "essentialai/rnj-1-instruct": { "id": "essentialai/rnj-1-instruct", "name": "EssentialAI: Rnj 1 Instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-05", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 6554 }, "cost": { "input": 0.15, "output": 0.15 } }, "deepseek/deepseek-r1-0528": { "id": "deepseek/deepseek-r1-0528", "name": "DeepSeek: R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-05-28", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.45, "output": 2.15, "cache_read": 0.2 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek: DeepSeek V4 Flash", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-24", "last_updated": "2026-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "deepseek/deepseek-v3.1-terminus": { "id": "deepseek/deepseek-v3.1-terminus", "name": "DeepSeek: DeepSeek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.21, "output": 0.79, "cache_read": 0.13 } }, "deepseek/deepseek-r1-distill-llama-70b": { "id": "deepseek/deepseek-r1-distill-llama-70b", "name": "DeepSeek: R1 Distill Llama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2025-01-23", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.7, "output": 0.8, "cache_read": 0.015 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek: DeepSeek V4 Pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "temperature": true, "release_date": "2026-04-24", "last_updated": "2026-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "deepseek/deepseek-r1": { "id": "deepseek/deepseek-r1", "name": "DeepSeek: R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 64000, "output": 16000 }, "cost": { "input": 0.7, "output": 2.5 } }, "deepseek/deepseek-v3.2-speciale": { "id": "deepseek/deepseek-v3.2-speciale", "name": "DeepSeek: DeepSeek V3.2 Speciale", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-12-01", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.4, "output": 1.2, "cache_read": 0.135 } }, "deepseek/deepseek-v3.2-exp": { "id": "deepseek/deepseek-v3.2-exp", "name": "DeepSeek: DeepSeek V3.2 Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.27, "output": 0.41 } }, "deepseek/deepseek-chat-v3-0324": { "id": "deepseek/deepseek-chat-v3-0324", "name": "DeepSeek: DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-03-24", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.2, "output": 0.77, "cache_read": 0.095 } }, "deepseek/deepseek-r1-distill-qwen-32b": { "id": "deepseek/deepseek-r1-distill-qwen-32b", "name": "DeepSeek: R1 Distill Qwen 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2025-01-01", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.29, "output": 0.29 } }, "deepseek/deepseek-chat": { "id": "deepseek/deepseek-chat", "name": "DeepSeek: DeepSeek V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.32, "output": 0.89, "cache_read": 0.15 } }, "deepseek/deepseek-v3.2": { "id": "deepseek/deepseek-v3.2", "name": "DeepSeek: DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-12-01", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.26, "output": 0.38, "cache_read": 0.125 } }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", "name": "DeepSeek: DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 7168 }, "cost": { "input": 0.15, "output": 0.75 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax: MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 }, "cost": { "input": 0.25, "output": 1.2, "cache_read": 0.029 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "MiniMax: MiniMax M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 39322 }, "cost": { "input": 0.27, "output": 0.95, "cache_read": 0.03 } }, "minimax/minimax-01": { "id": "minimax/minimax-01", "name": "MiniMax: MiniMax-01", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-01-15", "last_updated": "2025-01-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000192, "output": 1000192 }, "cost": { "input": 0.2, "output": 1.1 } }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", "name": "MiniMax: MiniMax M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 512000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax: MiniMax M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-23", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 196608 }, "cost": { "input": 0.255, "output": 1, "cache_read": 0.03 } }, "minimax/minimax-m2-her": { "id": "minimax/minimax-m2-her", "name": "MiniMax: MiniMax M2-her", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-23", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 2048 }, "cost": { "input": 0.3, "output": 1.2 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax: MiniMax M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax-m2.7", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m1": { "id": "minimax/minimax-m1", "name": "MiniMax: MiniMax M1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 40000 }, "cost": { "input": 0.4, "output": 2.2 } }, "stealth/claude-opus-4.7": { "id": "stealth/claude-opus-4.7", "name": "Stealth: Claude Opus 4.7 (20% off)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-05-27", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 4, "output": 20, "cache_read": 0.4, "cache_write": 5 } }, "stealth/claude-sonnet-4.6": { "id": "stealth/claude-sonnet-4.6", "name": "Stealth: Claude Sonnet 4.6 (20% off)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-17", "last_updated": "2026-05-27", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 2.4, "output": 12, "cache_read": 0.24, "cache_write": 3 } }, "stealth/claude-opus-4.6": { "id": "stealth/claude-opus-4.6", "name": "Stealth: Claude Opus 4.6 (20% off)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": true, "release_date": "2026-02-05", "last_updated": "2026-05-27", "modalities": { "input": ["image", "pdf", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 4, "output": 20, "cache_read": 0.4, "cache_write": 5 } }, "kwaipilot/kat-coder-pro-v2": { "id": "kwaipilot/kat-coder-pro-v2", "name": "Kwaipilot: KAT-Coder-Pro V2", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 80000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "nousresearch/hermes-2-pro-llama-3-8b": { "id": "nousresearch/hermes-2-pro-llama-3-8b", "name": "NousResearch: Hermes 2 Pro - Llama-3 8B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-05-27", "last_updated": "2024-06-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.14, "output": 0.14 } }, "nousresearch/hermes-4-405b": { "id": "nousresearch/hermes-4-405b", "name": "Nous: Hermes 4 405B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2025-08-25", "last_updated": "2025-08-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 26215 }, "cost": { "input": 1, "output": 3 } }, "nousresearch/hermes-3-llama-3.1-70b": { "id": "nousresearch/hermes-3-llama-3.1-70b", "name": "Nous: Hermes 3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-18", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.3, "output": 0.3 } }, "nousresearch/hermes-3-llama-3.1-405b": { "id": "nousresearch/hermes-3-llama-3.1-405b", "name": "Nous: Hermes 3 405B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-16", "last_updated": "2024-08-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 1, "output": 1 } }, "nousresearch/hermes-4-70b": { "id": "nousresearch/hermes-4-70b", "name": "Nous: Hermes 4 70B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2025-08-25", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.13, "output": 0.4, "cache_read": 0.055 } } } }, "lucidquery": { "id": "lucidquery", "env": ["LUCIDQUERY_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.lucidquery.com/v1", "name": "LucidQuery", "doc": "https://lucidquery.com/docs", "models": { "lucidnova-rf1-100b": { "id": "lucidnova-rf1-100b", "name": "LucidNova RF1 100B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "nova", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2025-09-16", "release_date": "2024-12-28", "last_updated": "2025-09-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 120000, "output": 8000 }, "cost": { "input": 2, "output": 5 } }, "lucidquery-nexus-coder": { "id": "lucidquery-nexus-coder", "name": "LucidQuery Nexus Coder", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "lucid", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2025-08-01", "release_date": "2025-09-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 250000, "output": 60000 }, "cost": { "input": 2, "output": 5 } }, "lucidquery-agi-01-swift": { "id": "lucidquery-agi-01-swift", "name": "AGI-01 Swift", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "agi", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2026-06-05", "release_date": "2026-06-16", "last_updated": "2026-06-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 120000 }, "cost": { "input": 2.5, "output": 15 } }, "lucidquery-agi-01-frontier": { "id": "lucidquery-agi-01-frontier", "name": "AGI-01 Frontier", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "agi", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2026-06-05", "release_date": "2026-06-16", "last_updated": "2026-06-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 120000 }, "cost": { "input": 4.5, "output": 22 } } } }, "meganova": { "id": "meganova", "env": ["MEGANOVA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.meganova.ai/v1", "name": "Meganova", "doc": "https://docs.meganova.ai", "models": { "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.1, "output": 0.3 } }, "moonshotai/Kimi-K2-Thinking": { "id": "moonshotai/Kimi-K2-Thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.6 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2026-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.45, "output": 2.8 } }, "Qwen/Qwen2.5-VL-32B-Instruct": { "id": "Qwen/Qwen2.5-VL-32B-Instruct", "name": "Qwen2.5 VL 32B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.2, "output": 0.6 } }, "Qwen/Qwen3.5-Plus": { "id": "Qwen/Qwen3.5-Plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02", "last_updated": "2026-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.4, "output": 2.4, "reasoning": 2.4 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.09, "output": 0.6 } }, "XiaomiMiMo/MiMo-V2-Flash": { "id": "XiaomiMiMo/MiMo-V2-Flash", "name": "MiMo V2 Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32000 }, "cost": { "input": 0.1, "output": 0.3 } }, "mistralai/Mistral-Small-3.2-24B-Instruct-2506": { "id": "mistralai/Mistral-Small-3.2-24B-Instruct-2506", "name": "Mistral Small 3.2 24B Instruct", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mistralai/Mistral-Nemo-Instruct-2407": { "id": "mistralai/Mistral-Nemo-Instruct-2407", "name": "Mistral Nemo Instruct 2407", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.02, "output": 0.04 } }, "zai-org/GLM-4.6": { "id": "zai-org/GLM-4.6", "name": "GLM-4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.45, "output": 1.9 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.8, "output": 2.56 } }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.2, "output": 0.8 } }, "deepseek-ai/DeepSeek-V3-0324": { "id": "deepseek-ai/DeepSeek-V3-0324", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.25, "output": 0.88 } }, "deepseek-ai/DeepSeek-R1-0528": { "id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-07", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 64000 }, "cost": { "input": 0.5, "output": 2.15 } }, "deepseek-ai/DeepSeek-V3.1": { "id": "deepseek-ai/DeepSeek-V3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-25", "last_updated": "2025-08-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 1 } }, "deepseek-ai/DeepSeek-V3.2-Exp": { "id": "deepseek-ai/DeepSeek-V3.2-Exp", "name": "DeepSeek V3.2 Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-10", "last_updated": "2025-10-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.27, "output": 0.4 } }, "deepseek-ai/DeepSeek-V3.2": { "id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-03", "last_updated": "2025-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 164000, "output": 164000 }, "cost": { "input": 0.26, "output": 0.38 } }, "MiniMaxAI/MiniMax-M2.1": { "id": "MiniMaxAI/MiniMax-M2.1", "name": "MiniMax M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 131072 }, "cost": { "input": 0.28, "output": 1.2 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } } } }, "inferx": { "id": "inferx", "env": ["INFERX_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://model.inferx.net/v1", "name": "InferX", "doc": "https://model.inferx.net/endpoints", "models": { "google/gemma-4-31b-it-fp8": { "id": "google/gemma-4-31b-it-fp8", "name": "Gemma 4 31B IT FP8", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3-coder-next-fp8-1m": { "id": "qwen/qwen3-coder-next-fp8-1m", "name": "Qwen3 Coder Next FP8 1M", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1024000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3.5-122b-a10b-nvfp4": { "id": "qwen/qwen3.5-122b-a10b-nvfp4", "name": "Qwen3.5 122B A10B NVFP4", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256144, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3.6-35b-a3b-fp8": { "id": "qwen/qwen3.6-35b-a3b-fp8", "name": "Qwen3.6 35B A3B FP8", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3-coder-next-fp8": { "id": "qwen/qwen3-coder-next-fp8", "name": "Qwen3 Coder Next FP8", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256144, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3.6-27b-fp8": { "id": "qwen/qwen3.6-27b-fp8", "name": "Qwen3.6 27B FP8", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0, "output": 0 } } } }, "perplexity": { "id": "perplexity", "env": ["PERPLEXITY_API_KEY"], "npm": "@ai-sdk/perplexity", "name": "Perplexity", "doc": "https://docs.perplexity.ai", "models": { "sonar-reasoning-pro": { "id": "sonar-reasoning-pro", "name": "Sonar Reasoning Pro", "description": "Web-grounded Sonar for multi-step research questions that need cited reasoning", "family": "sonar-reasoning", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 2, "output": 8 } }, "sonar": { "id": "sonar", "name": "Sonar", "description": "Fast web-grounded Sonar for current answers, citations, and lightweight retrieval", "family": "sonar", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 1, "output": 1 } }, "sonar-pro": { "id": "sonar-pro", "name": "Sonar Pro", "description": "Deeper Sonar search model with broader retrieval and stronger synthesis", "family": "sonar-pro", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-09-01", "release_date": "2024-01-01", "last_updated": "2025-09-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "cost": { "input": 3, "output": 15 } }, "sonar-deep-research": { "id": "sonar-deep-research", "name": "Perplexity Sonar Deep Research", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "temperature": false, "knowledge": "2025-01", "release_date": "2025-02-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 2, "output": 8, "reasoning": 3 } } } }, "amazon-bedrock": { "id": "amazon-bedrock", "env": ["AWS_ACCESS_KEY_ID", "AWS_SECRET_ACCESS_KEY", "AWS_REGION", "AWS_BEARER_TOKEN_BEDROCK"], "npm": "@ai-sdk/amazon-bedrock", "name": "Amazon Bedrock", "doc": "https://docs.aws.amazon.com/bedrock/latest/userguide/models-supported.html", "models": { "global.anthropic.claude-haiku-4-5-20251001-v1:0": { "id": "global.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "Claude Haiku 4.5 (Global)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "global.anthropic.claude-sonnet-4-5-20250929-v1:0": { "id": "global.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "Claude Sonnet 4.5 (Global)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "jp.anthropic.claude-haiku-4-5-20251001-v1:0": { "id": "jp.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "Claude Haiku 4.5 (JP)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "us.meta.llama4-scout-17b-instruct-v1:0": { "id": "us.meta.llama4-scout-17b-instruct-v1:0", "name": "Llama 4 Scout 17B Instruct (US)", "description": "Open Llama with long-context vision for efficient multimodal agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 3500000, "output": 16384 }, "cost": { "input": 0.17, "output": 0.66 } }, "minimax.minimax-m2": { "id": "minimax.minimax-m2", "name": "MiniMax M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204608, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2 } }, "anthropic.claude-opus-4-7": { "id": "anthropic.claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "eu.anthropic.claude-sonnet-4-6": { "id": "eu.anthropic.claude-sonnet-4-6", "name": "Claude Sonnet 4.6 (EU)", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3.3, "output": 16.5, "cache_read": 0.33, "cache_write": 4.125 } }, "mistral.voxtral-small-24b-2507": { "id": "mistral.voxtral-small-24b-2507", "name": "Voxtral Small 24B 2507", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 8192 }, "cost": { "input": 0.15, "output": 0.35 } }, "mistral.ministral-3-3b-instruct": { "id": "mistral.ministral-3-3b-instruct", "name": "Ministral 3 3B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.1, "output": 0.1 } }, "openai.gpt-oss-20b": { "id": "openai.gpt-oss-20b", "name": "gpt-oss-20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "provider": { "npm": "@ai-sdk/amazon-bedrock/mantle", "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/v1", "shape": "responses" }, "cost": { "input": 0.07, "output": 0.3 } }, "anthropic.claude-opus-4-6-v1": { "id": "anthropic.claude-opus-4-6-v1", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "openai.gpt-oss-safeguard-20b": { "id": "openai.gpt-oss-safeguard-20b", "name": "GPT OSS Safeguard 20B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-29", "last_updated": "2025-10-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.07, "output": 0.2 } }, "anthropic.claude-opus-4-5-20251101-v1:0": { "id": "anthropic.claude-opus-4-5-20251101-v1:0", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-08-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "global.anthropic.claude-fable-5": { "id": "global.anthropic.claude-fable-5", "name": "Claude Fable 5 (Global)", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "openai.gpt-oss-120b-1:0": { "id": "openai.gpt-oss-120b-1:0", "name": "gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "anthropic.claude-sonnet-4-5-20250929-v1:0": { "id": "anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "amazon.nova-pro-v1:0": { "id": "amazon.nova-pro-v1:0", "name": "Nova Pro", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova-pro", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 8192 }, "cost": { "input": 0.8, "output": 3.2, "cache_read": 0.2 } }, "qwen.qwen3-coder-next": { "id": "qwen.qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-06", "last_updated": "2026-02-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.22, "output": 1.8 } }, "us.anthropic.claude-opus-4-7": { "id": "us.anthropic.claude-opus-4-7", "name": "Claude Opus 4.7 (US)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "nvidia.nemotron-nano-9b-v2": { "id": "nvidia.nemotron-nano-9b-v2", "name": "NVIDIA Nemotron Nano 9B v2", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.06, "output": 0.23 } }, "qwen.qwen3-32b-v1:0": { "id": "qwen.qwen3-32b-v1:0", "name": "Qwen3 32B (dense)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-18", "last_updated": "2025-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "jp.anthropic.claude-sonnet-4-6": { "id": "jp.anthropic.claude-sonnet-4-6", "name": "Claude Sonnet 4.6 (JP)", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "deepseek.r1-v1:0": { "id": "deepseek.r1-v1:0", "name": "DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 1.35, "output": 5.4 } }, "mistral.mistral-large-3-675b-instruct": { "id": "mistral.mistral-large-3-675b-instruct", "name": "Mistral Large 3", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.5, "output": 1.5 } }, "google.gemma-3-27b-it": { "id": "google.gemma-3-27b-it", "name": "Google Gemma 3 27B Instruct", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-27", "last_updated": "2025-07-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 8192 }, "cost": { "input": 0.12, "output": 0.2 } }, "anthropic.claude-sonnet-4-6": { "id": "anthropic.claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "amazon.nova-2-lite-v1:0": { "id": "amazon.nova-2-lite-v1:0", "name": "Nova 2 Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "nova", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.33, "output": 2.75 } }, "openai.gpt-oss-safeguard-120b": { "id": "openai.gpt-oss-safeguard-120b", "name": "GPT OSS Safeguard 120B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-29", "last_updated": "2025-10-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "mistral.ministral-3-8b-instruct": { "id": "mistral.ministral-3-8b-instruct", "name": "Ministral 3 8B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15 } }, "eu.anthropic.claude-opus-4-6-v1": { "id": "eu.anthropic.claude-opus-4-6-v1", "name": "Claude Opus 4.6 (EU)", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5.5, "output": 27.5, "cache_read": 0.55, "cache_write": 6.875 } }, "au.anthropic.claude-opus-4-6-v1": { "id": "au.anthropic.claude-opus-4-6-v1", "name": "AU Anthropic Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 16.5, "output": 82.5, "cache_read": 1.65, "cache_write": 20.625 } }, "jp.anthropic.claude-sonnet-5": { "id": "jp.anthropic.claude-sonnet-5", "name": "Claude Sonnet 5 (JP)", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "openai.gpt-oss-120b": { "id": "openai.gpt-oss-120b", "name": "gpt-oss-120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "provider": { "npm": "@ai-sdk/amazon-bedrock/mantle", "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/v1", "shape": "responses" }, "cost": { "input": 0.15, "output": 0.6 } }, "global.anthropic.claude-sonnet-5": { "id": "global.anthropic.claude-sonnet-5", "name": "Claude Sonnet 5 (Global)", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "global.anthropic.claude-opus-4-8": { "id": "global.anthropic.claude-opus-4-8", "name": "Claude Opus 4.8 (Global)", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "us.meta.llama4-maverick-17b-instruct-v1:0": { "id": "us.meta.llama4-maverick-17b-instruct-v1:0", "name": "Llama 4 Maverick 17B Instruct (US)", "description": "Open multimodal Llama for strong reasoning with efficient everyday serving", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.24, "output": 0.97 } }, "openai.gpt-5.4": { "id": "openai.gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/amazon-bedrock/mantle", "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1", "shape": "responses" }, "cost": { "input": 2.75, "output": 16.5, "cache_read": 0.275 } }, "anthropic.claude-sonnet-5": { "id": "anthropic.claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "mistral.devstral-2-123b": { "id": "mistral.devstral-2-123b", "name": "Devstral 2 123B", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 0.4, "output": 2 } }, "zai.glm-4.7": { "id": "zai.glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2 } }, "anthropic.claude-fable-5": { "id": "anthropic.claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "writer.palmyra-x4-v1:0": { "id": "writer.palmyra-x4-v1:0", "name": "Palmyra X4", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "palmyra", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 122880, "output": 8192 }, "cost": { "input": 2.5, "output": 10 } }, "mistral.magistral-small-2509": { "id": "mistral.magistral-small-2509", "name": "Magistral Small 1.2", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 40000 }, "cost": { "input": 0.5, "output": 1.5 } }, "qwen.qwen3-coder-480b-a35b-v1:0": { "id": "qwen.qwen3-coder-480b-a35b-v1:0", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-18", "last_updated": "2025-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.22, "output": 1.8 } }, "amazon.nova-micro-v1:0": { "id": "amazon.nova-micro-v1:0", "name": "Nova Micro", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-micro", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.035, "output": 0.14, "cache_read": 0.00875 } }, "mistral.pixtral-large-2502-v1:0": { "id": "mistral.pixtral-large-2502-v1:0", "name": "Pixtral Large (25.02)", "description": "Mistral vision-language model for image understanding and multimodal chat", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-08", "last_updated": "2025-04-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 2, "output": 6 } }, "us.anthropic.claude-opus-4-6-v1": { "id": "us.anthropic.claude-opus-4-6-v1", "name": "Claude Opus 4.6 (US)", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "jp.anthropic.claude-opus-4-7": { "id": "jp.anthropic.claude-opus-4-7", "name": "Claude Opus 4.7 (JP)", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "au.anthropic.claude-sonnet-4-5-20250929-v1:0": { "id": "au.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "Claude Sonnet 4.5 (AU)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "deepseek.v3-v1:0": { "id": "deepseek.v3-v1:0", "name": "DeepSeek-V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-09-18", "last_updated": "2025-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 81920 }, "cost": { "input": 0.58, "output": 1.68 } }, "anthropic.claude-opus-4-1-20250805-v1:0": { "id": "anthropic.claude-opus-4-1-20250805-v1:0", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "jp.anthropic.claude-sonnet-4-5-20250929-v1:0": { "id": "jp.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "Claude Sonnet 4.5 (JP)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "google.gemma-3-4b-it": { "id": "google.gemma-3-4b-it", "name": "Gemma 3 4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.04, "output": 0.08 } }, "eu.anthropic.claude-sonnet-4-5-20250929-v1:0": { "id": "eu.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "Claude Sonnet 4.5 (EU)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3.3, "output": 16.5, "cache_read": 0.33, "cache_write": 4.125 } }, "qwen.qwen3-vl-235b-a22b": { "id": "qwen.qwen3-vl-235b-a22b", "name": "Qwen/Qwen3-VL-235B-A22B-Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-04", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.3, "output": 1.5 } }, "writer.palmyra-x5-v1:0": { "id": "writer.palmyra-x5-v1:0", "name": "Palmyra X5", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "palmyra", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1040000, "output": 8192 }, "cost": { "input": 0.6, "output": 6 } }, "us.anthropic.claude-sonnet-4-6": { "id": "us.anthropic.claude-sonnet-4-6", "name": "Claude Sonnet 4.6 (US)", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "au.anthropic.claude-haiku-4-5-20251001-v1:0": { "id": "au.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "Claude Haiku 4.5 (AU)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "meta.llama3-3-70b-instruct-v1:0": { "id": "meta.llama3-3-70b-instruct-v1:0", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.72, "output": 0.72 } }, "zai.glm-5": { "id": "zai.glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 101376 }, "cost": { "input": 1, "output": 3.2 } }, "us.anthropic.claude-opus-4-8": { "id": "us.anthropic.claude-opus-4-8", "name": "Claude Opus 4.8 (US)", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "global.anthropic.claude-opus-4-7": { "id": "global.anthropic.claude-opus-4-7", "name": "Claude Opus 4.7 (Global)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "us.anthropic.claude-fable-5": { "id": "us.anthropic.claude-fable-5", "name": "Claude Fable 5 (US)", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "amazon.nova-lite-v1:0": { "id": "amazon.nova-lite-v1:0", "name": "Nova Lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-lite", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 8192 }, "cost": { "input": 0.06, "output": 0.24, "cache_read": 0.015 } }, "us.anthropic.claude-haiku-4-5-20251001-v1:0": { "id": "us.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "Claude Haiku 4.5 (US)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "xai.grok-4.3": { "id": "xai.grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-06-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 131072 }, "provider": { "npm": "@ai-sdk/amazon-bedrock/mantle", "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1", "shape": "responses" }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "mistral.voxtral-mini-3b-2507": { "id": "mistral.voxtral-mini-3b-2507", "name": "Voxtral Mini 3B 2507", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["audio", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.04, "output": 0.04 } }, "moonshot.kimi-k2-thinking": { "id": "moonshot.kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262143, "output": 16000 }, "cost": { "input": 0.6, "output": 2.5 } }, "meta.llama3-1-70b-instruct-v1:0": { "id": "meta.llama3-1-70b-instruct-v1:0", "name": "Llama 3.1 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.72, "output": 0.72 } }, "us.deepseek.r1-v1:0": { "id": "us.deepseek.r1-v1:0", "name": "DeepSeek-R1 (US)", "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 1.35, "output": 5.4 } }, "global.anthropic.claude-sonnet-4-6": { "id": "global.anthropic.claude-sonnet-4-6", "name": "Claude Sonnet 4.6 (Global)", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic.claude-haiku-4-5-20251001-v1:0": { "id": "anthropic.claude-haiku-4-5-20251001-v1:0", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "moonshotai.kimi-k2.5": { "id": "moonshotai.kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "release_date": "2026-02-06", "last_updated": "2026-02-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262143, "output": 16000 }, "cost": { "input": 0.6, "output": 3 } }, "au.anthropic.claude-opus-4-8": { "id": "au.anthropic.claude-opus-4-8", "name": "Claude Opus 4.8 (AU)", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "nvidia.nemotron-nano-12b-v2": { "id": "nvidia.nemotron-nano-12b-v2", "name": "NVIDIA Nemotron Nano 12B v2 VL BF16", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.2, "output": 0.6 } }, "zai.glm-4.7-flash": { "id": "zai.glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0.07, "output": 0.4 } }, "meta.llama4-scout-17b-instruct-v1:0": { "id": "meta.llama4-scout-17b-instruct-v1:0", "name": "Llama 4 Scout 17B Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 3500000, "output": 16384 }, "cost": { "input": 0.17, "output": 0.66 } }, "au.anthropic.claude-sonnet-5": { "id": "au.anthropic.claude-sonnet-5", "name": "Claude Sonnet 5 (AU)", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "qwen.qwen3-235b-a22b-2507-v1:0": { "id": "qwen.qwen3-235b-a22b-2507-v1:0", "name": "Qwen3 235B A22B 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-18", "last_updated": "2025-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.22, "output": 0.88 } }, "eu.anthropic.claude-haiku-4-5-20251001-v1:0": { "id": "eu.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "Claude Haiku 4.5 (EU)", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1.1, "output": 5.5, "cache_read": 0.11, "cache_write": 1.375 } }, "openai.gpt-oss-20b-1:0": { "id": "openai.gpt-oss-20b-1:0", "name": "gpt-oss-20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.07, "output": 0.3 } }, "jp.anthropic.claude-opus-4-8": { "id": "jp.anthropic.claude-opus-4-8", "name": "Claude Opus 4.8 (JP)", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic.claude-opus-4-8": { "id": "anthropic.claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "qwen.qwen3-coder-30b-a3b-v1:0": { "id": "qwen.qwen3-coder-30b-a3b-v1:0", "name": "Qwen3 Coder 30B A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-18", "last_updated": "2025-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.15, "output": 0.6 } }, "qwen.qwen3-next-80b-a3b": { "id": "qwen.qwen3-next-80b-a3b", "name": "Qwen/Qwen3-Next-80B-A3B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-18", "last_updated": "2025-11-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.14, "output": 1.4 } }, "openai.gpt-5.5": { "id": "openai.gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/amazon-bedrock/mantle", "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1", "shape": "responses" }, "cost": { "input": 5.5, "output": 33, "cache_read": 0.55 } }, "au.anthropic.claude-sonnet-4-6": { "id": "au.anthropic.claude-sonnet-4-6", "name": "AU Anthropic Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3.3, "output": 16.5, "cache_read": 0.33, "cache_write": 4.125 } }, "us.anthropic.claude-sonnet-5": { "id": "us.anthropic.claude-sonnet-5", "name": "Claude Sonnet 5 (US)", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "us.anthropic.claude-opus-4-5-20251101-v1:0": { "id": "us.anthropic.claude-opus-4-5-20251101-v1:0", "name": "Claude Opus 4.5 (US)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-08-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "minimax.minimax-m2.5": { "id": "minimax.minimax-m2.5", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 98304 }, "cost": { "input": 0.3, "output": 1.2 } }, "eu.anthropic.claude-opus-4-8": { "id": "eu.anthropic.claude-opus-4-8", "name": "Claude Opus 4.8 (EU)", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5.5, "output": 27.5, "cache_read": 0.55, "cache_write": 6.875 } }, "eu.anthropic.claude-opus-4-7": { "id": "eu.anthropic.claude-opus-4-7", "name": "Claude Opus 4.7 (EU)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5.5, "output": 27.5, "cache_read": 0.55, "cache_write": 6.875 } }, "meta.llama3-1-8b-instruct-v1:0": { "id": "meta.llama3-1-8b-instruct-v1:0", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.22, "output": 0.22 } }, "us.anthropic.claude-opus-4-1-20250805-v1:0": { "id": "us.anthropic.claude-opus-4-1-20250805-v1:0", "name": "Claude Opus 4.1 (US)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "meta.llama4-maverick-17b-instruct-v1:0": { "id": "meta.llama4-maverick-17b-instruct-v1:0", "name": "Llama 4 Maverick 17B Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.24, "output": 0.97 } }, "global.anthropic.claude-opus-4-5-20251101-v1:0": { "id": "global.anthropic.claude-opus-4-5-20251101-v1:0", "name": "Claude Opus 4.5 (Global)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-08-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "nvidia.nemotron-super-3-120b": { "id": "nvidia.nemotron-super-3-120b", "name": "NVIDIA Nemotron 3 Super 120B A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.15, "output": 0.65 } }, "eu.anthropic.claude-sonnet-5": { "id": "eu.anthropic.claude-sonnet-5", "name": "Claude Sonnet 5 (EU)", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2.2, "output": 11, "cache_read": 0.22, "cache_write": 2.75 } }, "nvidia.nemotron-nano-3-30b": { "id": "nvidia.nemotron-nano-3-30b", "name": "NVIDIA Nemotron Nano 3 30B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.06, "output": 0.24 } }, "eu.anthropic.claude-opus-4-5-20251101-v1:0": { "id": "eu.anthropic.claude-opus-4-5-20251101-v1:0", "name": "Claude Opus 4.5 (EU)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-08-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5.5, "output": 27.5, "cache_read": 0.55, "cache_write": 6.875 } }, "global.anthropic.claude-opus-4-6-v1": { "id": "global.anthropic.claude-opus-4-6-v1", "name": "Claude Opus 4.6 (Global)", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "us.anthropic.claude-sonnet-4-5-20250929-v1:0": { "id": "us.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "Claude Sonnet 4.5 (US)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "google.gemma-3-12b-it": { "id": "google.gemma-3-12b-it", "name": "Google Gemma 3 12B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.049999999999999996, "output": 0.09999999999999999 } }, "minimax.minimax-m2.1": { "id": "minimax.minimax-m2.1", "name": "MiniMax M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "eu.anthropic.claude-fable-5": { "id": "eu.anthropic.claude-fable-5", "name": "Claude Fable 5 (EU)", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 11, "output": 55, "cache_read": 1.1, "cache_write": 13.75 } }, "deepseek.v3.2": { "id": "deepseek.v3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2026-02-06", "last_updated": "2026-02-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 81920 }, "cost": { "input": 0.62, "output": 1.85 } }, "mistral.ministral-3-14b-instruct": { "id": "mistral.ministral-3-14b-instruct", "name": "Ministral 14B 3.0", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.2, "output": 0.2 } } } }, "umans-ai": { "id": "umans-ai", "env": ["UMANS_AI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.code.umans.ai/v1", "name": "Umans AI", "doc": "https://app.umans.ai/offers/code/docs/orgs", "models": { "umans-kimi-k2.7": { "id": "umans-kimi-k2.7", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "umans-glm-5.1": { "id": "umans-glm-5.1", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.29 } }, "umans-coder": { "id": "umans-coder", "name": "Umans Coder", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "umans-flash": { "id": "umans-flash", "name": "Umans Flash", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.15, "output": 1, "cache_read": 0.05 } }, "umans-glm-5.2": { "id": "umans-glm-5.2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 405504, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } } } }, "togetherai": { "id": "togetherai", "env": ["TOGETHER_API_KEY"], "npm": "@ai-sdk/togetherai", "name": "Together AI", "doc": "https://docs.together.ai/docs/serverless-models", "models": { "LiquidAI/LFM2-24B-A2B": { "id": "LiquidAI/LFM2-24B-A2B", "name": "LFM2-24B-A2B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "liquid", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-02-25", "last_updated": "2026-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.03, "output": 0.12 } }, "meta-llama/Meta-Llama-3-8B-Instruct-Lite": { "id": "meta-llama/Meta-Llama-3-8B-Instruct-Lite", "name": "Meta Llama 3 8B Instruct Lite", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.14, "output": 0.14 } }, "meta-llama/Llama-3.3-70B-Instruct-Turbo": { "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo", "name": "Llama 3.3 70B", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 1.04, "output": 1.04 } }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131000 }, "cost": { "input": 1.2, "output": 4.5, "cache_read": 0.2 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi K2.5", "description": "Legacy model retained for compatibility with older integrations", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.5, "output": 2.8 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Kimi coding model for software agents, refactors, and repository reasoning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-14", "last_updated": "2026-06-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B Instruct", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.39, "output": 0.97 } }, "google/gemma-3n-E4B-it": { "id": "google/gemma-3n-E4B-it", "name": "Gemma 3N E4B Instruct", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.06, "output": 0.12 } }, "Qwen/Qwen3.7-Max": { "id": "Qwen/Qwen3.7-Max", "name": "Qwen3.7 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 500000 }, "cost": { "input": 1.25, "output": 3.75 } }, "Qwen/Qwen3.6-Plus": { "id": "Qwen/Qwen3.6-Plus", "name": "Qwen3.6 Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 500000 }, "cost": { "input": 0.5, "output": 3 } }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen3.5 397B A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-02-16", "last_updated": "2026-06-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 130000 }, "cost": { "input": 0.6, "output": 3.6 } }, "Qwen/Qwen3-Coder-Next-FP8": { "id": "Qwen/Qwen3-Coder-Next-FP8", "name": "Qwen3 Coder Next FP8", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2026-02-03", "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.5, "output": 1.2 } }, "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Legacy model retained for compatibility with older integrations", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 2, "output": 2 } }, "Qwen/Qwen3-235B-A22B-Instruct-2507-tput": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507-tput", "name": "Qwen3 235B A22B Instruct 2507 FP8", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2, "output": 0.6 } }, "Qwen/Qwen2.5-7B-Instruct-Turbo": { "id": "Qwen/Qwen2.5-7B-Instruct-Turbo", "name": "Qwen 2.5 7B Instruct Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.3, "output": 0.3 } }, "Qwen/Qwen3.5-9B": { "id": "Qwen/Qwen3.5-9B", "name": "Qwen3.5 9B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.17, "output": 0.25 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.15, "output": 0.6 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.2 } }, "pearl-ai/gemma-4-31b-it": { "id": "pearl-ai/gemma-4-31b-it", "name": "Pearl AI Gemma 4 31B Instruct", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.28, "output": 0.86 } }, "nvidia/nemotron-3-ultra-550b-a55b": { "id": "nvidia/nemotron-3-ultra-550b-a55b", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512300, "output": 512300 }, "cost": { "input": 0.6, "output": 3.6, "cache_read": 0.2 } }, "deepcogito/cogito-v2-1-671b": { "id": "deepcogito/cogito-v2-1-671b", "name": "Cogito v2.1 671B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "cogito", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "temperature": true, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 1.25, "output": 1.25 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1, "output": 3.2 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-16", "last_updated": "2026-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 164000 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-11", "release_date": "2026-04-07", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4 } }, "deepseek-ai/DeepSeek-R1": { "id": "deepseek-ai/DeepSeek-R1", "name": "DeepSeek-R1", "description": "Legacy model retained for compatibility with older integrations", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163839, "output": 163839 }, "status": "deprecated", "cost": { "input": 3, "output": 7 } }, "deepseek-ai/DeepSeek-V3-1": { "id": "deepseek-ai/DeepSeek-V3-1", "name": "DeepSeek V3.1", "description": "Legacy model retained for compatibility with older integrations", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "status": "deprecated", "cost": { "input": 0.6, "output": 1.7 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512000, "output": 384000 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.2 } }, "deepseek-ai/DeepSeek-V3": { "id": "deepseek-ai/DeepSeek-V3", "name": "DeepSeek-V3", "description": "Legacy model retained for compatibility with older integrations", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-12-26", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "status": "deprecated", "cost": { "input": 1.25, "output": 1.25 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "status": "deprecated", "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "MiniMaxAI/MiniMax-M3": { "id": "MiniMaxAI/MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 250000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "MiniMaxAI/MiniMax-M2.7": { "id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "essentialai/Rnj-1-Instruct": { "id": "essentialai/Rnj-1-Instruct", "name": "Rnj-1 Instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "rnj", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12-05", "last_updated": "2025-12-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.15, "output": 0.15 } } } }, "frogbot": { "id": "frogbot", "env": ["FROGBOT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://app.frogbot.ai/api/v1", "name": "FrogBot", "doc": "https://docs.frogbot.ai", "models": { "minimax-m2-5": { "id": "minimax-m2-5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2025-01-15", "last_updated": "2025-02-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 192000, "output": 8192 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi-K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "1970-01-01", "last_updated": "1970-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "zai-glm-5-1": { "id": "zai-glm-5-1", "name": "Z.AI GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-01-20", "last_updated": "2025-02-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 198000, "output": 8192 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "grok-code-fast-1": { "id": "grok-code-fast-1", "name": "Grok 4.1 Fast (Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.5, "cache_read": 0.02 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-03-20", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.31 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-07-17", "last_updated": "2025-07-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075 } }, "gpt-4o": { "id": "gpt-4o", "name": "GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "qwen-3-6-plus": { "id": "qwen-3-6-plus", "name": "Qwen 3.6 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-03", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.1 } }, "grok-4-3": { "id": "grok-4-3", "name": "Grok 4.3", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-11", "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek v4 Pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.14 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "grok-4-1-fast-non-reasoning": { "id": "grok-4-1-fast-non-reasoning", "name": "Grok 4.1 Fast (Non-Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 128000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "minimax-m2-7": { "id": "minimax-m2-7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 192000, "output": 8192 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "gpt-5-3-codex": { "id": "gpt-5-3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5-4-nano": { "id": "gpt-5-4-nano", "name": "GPT-5.4 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "gpt-5-4-mini": { "id": "gpt-5-4-mini", "name": "GPT-5.4 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "1970-01-01", "last_updated": "1970-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "1970-01-01", "last_updated": "1970-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6 } }, "gpt-5-5": { "id": "gpt-5-5", "name": "GPT-5.5", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gemini-3-1-pro-preview": { "id": "gemini-3-1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-02-18", "last_updated": "2026-02-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "grok-4-1-fast-reasoning": { "id": "grok-4-1-fast-reasoning", "name": "Grok 4.1 Fast (Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-11", "release_date": "2025-11-25", "last_updated": "2025-11-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 128000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "1970-01-01", "last_updated": "1970-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.07, "output": 0.2 } } } }, "openrouter": { "id": "openrouter", "env": ["OPENROUTER_API_KEY"], "npm": "@openrouter/ai-sdk-provider", "api": "https://openrouter.ai/api/v1", "name": "OpenRouter", "doc": "https://openrouter.ai/models", "models": { "inclusionai/ling-2.6-1t": { "id": "inclusionai/ling-2.6-1t", "name": "Ling-2.6-1T", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.075, "output": 0.625, "cache_read": 0.015 } }, "inclusionai/ring-2.6-1t": { "id": "inclusionai/ring-2.6-1t", "name": "Ring-2.6-1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "ring", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-05-08", "last_updated": "2026-05-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.075, "output": 0.625, "cache_read": 0.015 } }, "inclusionai/ling-2.6-flash": { "id": "inclusionai/ling-2.6-flash", "name": "Ling-2.6-flash", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.01, "output": 0.03, "cache_read": 0.002 } }, "ibm-granite/granite-4.0-h-micro": { "id": "ibm-granite/granite-4.0-h-micro", "name": "Granite 4.0 Micro", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "granite", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-10-20", "last_updated": "2025-10-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131000, "output": 131000 }, "cost": { "input": 0.017, "output": 0.112 } }, "ibm-granite/granite-4.1-8b": { "id": "ibm-granite/granite-4.1-8b", "name": "Granite 4.1 8B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "granite", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.1, "cache_read": 0.05 } }, "meta-llama/llama-3.1-8b-instruct": { "id": "meta-llama/llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.02, "output": 0.03 } }, "meta-llama/llama-3.1-70b-instruct": { "id": "meta-llama/llama-3.1-70b-instruct", "name": "Llama 3.1 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.4, "output": 0.4 } }, "meta-llama/llama-3.2-1b-instruct": { "id": "meta-llama/llama-3.2-1b-instruct", "name": "Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 60000 }, "cost": { "input": 0.027, "output": 0.201 } }, "meta-llama/llama-4-maverick": { "id": "meta-llama/llama-4-maverick", "name": "Llama 4 Maverick", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 16384 }, "cost": { "input": 0.2, "output": 0.8 } }, "meta-llama/llama-3.2-11b-vision-instruct": { "id": "meta-llama/llama-3.2-11b-vision-instruct", "name": "Llama 3.2 11B Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.345, "output": 0.345 } }, "meta-llama/llama-3.3-70b-instruct:free": { "id": "meta-llama/llama-3.3-70b-instruct:free", "name": "Llama 3.3 70B Instruct (free)", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.1, "output": 0.32 } }, "meta-llama/llama-3.2-3b-instruct:free": { "id": "meta-llama/llama-3.2-3b-instruct:free", "name": "Llama 3.2 3B Instruct (free)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "meta-llama/llama-guard-4-12b": { "id": "meta-llama/llama-guard-4-12b", "name": "Llama Guard 4 12B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-04-30", "last_updated": "2025-04-30", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 16384 }, "cost": { "input": 0.18, "output": 0.18 } }, "meta-llama/llama-4-scout": { "id": "meta-llama/llama-4-scout", "name": "Llama 4 Scout", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 10000000, "output": 16384 }, "cost": { "input": 0.1, "output": 0.3 } }, "meta-llama/llama-3.2-3b-instruct": { "id": "meta-llama/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.33 } }, "~anthropic/claude-haiku-latest": { "id": "~anthropic/claude-haiku-latest", "name": "Anthropic Claude Haiku Latest", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "~anthropic/claude-fable-latest": { "id": "~anthropic/claude-fable-latest", "name": "Claude Fable Latest", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "~anthropic/claude-sonnet-latest": { "id": "~anthropic/claude-sonnet-latest", "name": "Anthropic Claude Sonnet Latest", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "~anthropic/claude-opus-latest": { "id": "~anthropic/claude-opus-latest", "name": "Claude Opus Latest", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "moonshotai/kimi-k2": { "id": "moonshotai/kimi-k2", "name": "Kimi K2 0711", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-12-31", "release_date": "2025-07-11", "last_updated": "2025-07-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 100352 }, "cost": { "input": 0.57, "output": 2.3 } }, "moonshotai/kimi-k2.7-code": { "id": "moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.72, "output": 3.5, "cache_read": 0.15 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 100352 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 256000 }, "cost": { "input": 0.375, "output": 2.025, "cache_read": 0.203 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.66, "output": 3.41, "cache_read": 0.15 } }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12-31", "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 100352 }, "cost": { "input": 0.6, "output": 2.5 } }, "baidu/ernie-4.5-vl-424b-a47b": { "id": "baidu/ernie-4.5-vl-424b-a47b", "name": "ERNIE 4.5 VL 424B A47B ", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "ernie", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16000 }, "cost": { "input": 0.42, "output": 1.25 } }, "perceptron/perceptron-mk1": { "id": "perceptron/perceptron-mk1", "name": "Perceptron Mk1", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0.15, "output": 1.5 } }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "reasoning": 1.5, "cache_read": 0.025, "cache_write": 0.083333 } }, "google/gemini-3.1-flash-image": { "id": "google/gemini-3.1-flash-image", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 3 } }, "google/gemma-3n-e4b-it": { "id": "google/gemma-3n-e4b-it", "name": "Gemma 3n 4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.06, "output": 0.12 } }, "google/gemma-4-26b-a4b-it:free": { "id": "google/gemma-4-26b-a4b-it:free", "name": "Gemma 4 26B A4B (free)", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375 } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.3, "output": 2.5, "reasoning": 2.5, "cache_read": 0.03, "cache_write": 0.083333 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "reasoning": 9, "cache_read": 0.15, "cache_write": 0.083333 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.12, "output": 0.35, "cache_read": 0.09 } }, "google/lyria-3-clip-preview": { "id": "google/lyria-3-clip-preview", "name": "Lyria 3 Clip Preview", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "lyria", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-03-30", "last_updated": "2026-03-30", "modalities": { "input": ["text", "image"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "google/gemini-3.1-pro-preview-customtools": { "id": "google/gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "output": 65536 }, "cost": { "input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemini-3.1-flash-lite-image": { "id": "google/gemini-3.1-flash-lite-image", "name": "Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["image", "text"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 65536, "output": 66000 }, "cost": { "input": 0.25, "output": 1.5 } }, "google/gemini-3-pro-image-preview": { "id": "google/gemini-3-pro-image-preview", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 32768 }, "cost": { "input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375 } }, "google/gemini-2.5-flash-image": { "id": "google/gemini-2.5-flash-image", "name": "Nano Banana", "description": "Nano Banana image model for fast generation, edits, and character-consistent assets", "family": "gemini-flash", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "cache_write": 0.083333 } }, "google/gemini-2.5-flash-lite": { "id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.1, "output": 0.4, "reasoning": 0.4, "cache_read": 0.01, "cache_write": 0.083333 } }, "google/gemini-3.1-flash-image-preview": { "id": "google/gemini-3.1-flash-image-preview", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["image", "text"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 3 } }, "google/gemini-2.5-pro-preview-05-06": { "id": "google/gemini-2.5-pro-preview-05-06", "name": "Gemini 2.5 Pro Preview 05-06", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image", "pdf", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.06, "output": 0.33 } }, "google/gemini-2.5-pro-preview": { "id": "google/gemini-2.5-pro-preview", "name": "Gemini 2.5 Pro Preview 06-05", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["pdf", "image", "text", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375 } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.5, "output": 3, "reasoning": 3, "cache_read": 0.05, "cache_write": 0.083333 } }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", "name": "Gemma 3 12B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.05, "output": 0.15 } }, "google/gemma-3-4b-it": { "id": "google/gemma-3-4b-it", "name": "Gemma 3 4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.05, "output": 0.1 } }, "google/gemma-3-27b-it": { "id": "google/gemma-3-27b-it", "name": "Gemma 3 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.08, "output": 0.16 } }, "google/lyria-3-pro-preview": { "id": "google/lyria-3-pro-preview", "name": "Lyria 3 Pro Preview", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "lyria", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-03-30", "last_updated": "2026-03-30", "modalities": { "input": ["text", "image"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "google/gemma-4-31b-it:free": { "id": "google/gemma-4-31b-it:free", "name": "Gemma 4 31B (free)", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "google/gemma-2-27b-it": { "id": "google/gemma-2-27b-it", "name": "Gemma 2 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-07-13", "last_updated": "2024-07-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0.65, "output": 0.65 } }, "google/gemini-3-pro-image": { "id": "google/gemini-3-pro-image", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 32768 }, "cost": { "input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375 } }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "reasoning": 1.5, "cache_read": 0.025, "cache_write": 0.083333 } }, "liquid/lfm-2.5-1.2b-thinking:free": { "id": "liquid/lfm-2.5-1.2b-thinking:free", "name": "LFM2.5-1.2B-Thinking (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "liquid", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-01-20", "last_updated": "2026-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "liquid/lfm-2.5-1.2b-instruct:free": { "id": "liquid/lfm-2.5-1.2b-instruct:free", "name": "LFM2.5-1.2B-Instruct (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "liquid", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-01-20", "last_updated": "2026-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "x-ai/grok-4.20": { "id": "x-ai/grok-4.20", "name": "Grok 4.20", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-09-01", "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "x-ai/grok-4.5": { "id": "x-ai/grok-4.5", "name": "Grok 4.5", "description": "xAI's latest Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 500000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.5, "tiers": [ { "input": 4, "output": 12, "cache_read": 1, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 12, "cache_read": 1 } } }, "x-ai/grok-4.20-multi-agent": { "id": "x-ai/grok-4.20-multi-agent", "name": "Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "high"] } ], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-09-01", "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2 } }, "~google/gemini-pro-latest": { "id": "~google/gemini-pro-latest", "name": "Google Gemini Pro Latest", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["audio", "pdf", "image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375 } }, "~google/gemini-flash-latest": { "id": "~google/gemini-flash-latest", "name": "Google Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video", "pdf", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "reasoning": 9, "cache_read": 0.15, "cache_write": 0.083333 } }, "microsoft/phi-4": { "id": "microsoft/phi-4", "name": "Phi 4", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-01-10", "last_updated": "2025-01-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.07, "output": 0.14 } }, "microsoft/wizardlm-2-8x22b": { "id": "microsoft/wizardlm-2-8x22b", "name": "WizardLM-2 8x22B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-04-16", "last_updated": "2024-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 8000 }, "cost": { "input": 0.62, "output": 0.62 } }, "poolside/laguna-xs-2.1:free": { "id": "poolside/laguna-xs-2.1:free", "name": "Laguna XS 2.1 (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-02", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "poolside/laguna-m.1": { "id": "poolside/laguna-m.1", "name": "Laguna M.1", "description": "Poolside's flagship agentic coding model for long-horizon work", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.2, "output": 0.4, "cache_read": 0.1 } }, "poolside/laguna-xs-2.1": { "id": "poolside/laguna-xs-2.1", "name": "Laguna XS 2.1", "description": "Agentic coding model from Poolside in the XS size class for local deployment", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-02", "last_updated": "2026-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.06, "output": 0.12, "cache_read": 0.03 } }, "poolside/laguna-m.1:free": { "id": "poolside/laguna-m.1:free", "name": "Laguna M.1 (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "writer/palmyra-x5": { "id": "writer/palmyra-x5", "name": "Palmyra X5", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "palmyra", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01-21", "last_updated": "2026-01-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1040000, "output": 8192 }, "cost": { "input": 0.6, "output": 6 } }, "z-ai/glm-4.7": { "id": "z-ai/glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 0.4, "output": 1.75, "cache_read": 0.08 } }, "z-ai/glm-4.5v": { "id": "z-ai/glm-4.5v", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8, "cache_read": 0.11 } }, "z-ai/glm-4.5": { "id": "z-ai/glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "z-ai/glm-5.1": { "id": "z-ai/glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 128000 }, "cost": { "input": 0.966, "output": 3.036, "cache_read": 0.1794 } }, "z-ai/glm-4.6": { "id": "z-ai/glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 16384 }, "cost": { "input": 0.43, "output": 1.75, "cache_read": 0.08 } }, "z-ai/glm-5.2": { "id": "z-ai/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.42, "output": 1.32, "cache_read": 0.078 } }, "z-ai/glm-4.6v": { "id": "z-ai/glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9, "cache_read": 0.055 } }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "z-ai/glm-4.5-air": { "id": "z-ai/glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.13, "output": 0.85, "cache_read": 0.025 } }, "z-ai/glm-4.7-flash": { "id": "z-ai/glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0.06, "output": 0.4, "cache_read": 0.01 } }, "z-ai/glm-5": { "id": "z-ai/glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 128000 }, "cost": { "input": 0.6, "output": 1.92, "cache_read": 0.12 } }, "z-ai/glm-5-turbo": { "id": "z-ai/glm-5-turbo", "name": "GLM-5-Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": false, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "openai/gpt-4o-mini-2024-07-18": { "id": "openai/gpt-4o-mini-2024-07-18", "name": "GPT-4o-mini (2024-07-18)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "o-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-10-31", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "openai/gpt-oss-safeguard-20b": { "id": "openai/gpt-oss-safeguard-20b", "name": "gpt-oss-safeguard-20b", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-29", "last_updated": "2025-10-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.075, "output": 0.3, "cache_read": 0.0375 } }, "openai/gpt-3.5-turbo-instruct": { "id": "openai/gpt-3.5-turbo-instruct", "name": "GPT-3.5 Turbo Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2021-09-30", "release_date": "2023-09-28", "last_updated": "2023-09-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4095, "output": 4096 }, "cost": { "input": 1.5, "output": 2 } }, "openai/gpt-5.2-chat": { "id": "openai/gpt-5.2-chat", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-10", "last_updated": "2025-12-10", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.6-luna-pro": { "id": "openai/gpt-5.6-luna-pro", "name": "GPT-5.6 Luna Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25 } }, "openai/o3": { "id": "openai/o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5.6-sol-pro": { "id": "openai/gpt-5.6-sol-pro", "name": "GPT-5.6 Sol Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25 } }, "openai/o4-mini-high": { "id": "openai/o4-mini-high", "name": "o4 Mini High", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-06-30", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "openai/gpt-audio": { "id": "openai/gpt-audio", "name": "GPT Audio", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "GPT-5.2 Pro", "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "openai/gpt-4o-mini-search-preview": { "id": "openai/gpt-4o-mini-search-preview", "name": "GPT-4o-mini Search Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "o-mini", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": false, "knowledge": "2023-10-31", "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6 } }, "openai/gpt-5.6-terra-pro": { "id": "openai/gpt-5.6-terra-pro", "name": "GPT-5.6 Terra Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 3.125 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5-chat": { "id": "openai/gpt-5-chat", "name": "GPT-5 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-3.5-turbo": { "id": "openai/gpt-3.5-turbo", "name": "GPT-3.5-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2021-09-01", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5 } }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 15, "output": 120 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "openai/gpt-4": { "id": "openai/gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 4096 }, "cost": { "input": 30, "output": 60 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "openai/gpt-3.5-turbo-16k": { "id": "openai/gpt-3.5-turbo-16k", "name": "GPT-3.5 Turbo 16k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2021-09-30", "release_date": "2023-08-28", "last_updated": "2023-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "output": 4096 }, "cost": { "input": 3, "output": 4 } }, "openai/o3-pro": { "id": "openai/o3-pro", "name": "o3-pro", "description": "High-effort o3 tier for difficult technical reasoning and careful answers", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": { "input": ["text", "pdf", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 20, "output": 80 } }, "openai/gpt-5.1-chat": { "id": "openai/gpt-5.1-chat", "name": "GPT-5.1 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "openai/gpt-4o-2024-05-13": { "id": "openai/gpt-4o-2024-05-13", "name": "GPT-4o (2024-05-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 5, "output": 15 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "openai/gpt-5.3-chat": { "id": "openai/gpt-5.3-chat", "name": "GPT-5.3 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-3.5-turbo-0613": { "id": "openai/gpt-3.5-turbo-0613", "name": "GPT-3.5 Turbo (older v0613)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2021-09-30", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4095, "output": 4096 }, "cost": { "input": 1, "output": 2 } }, "openai/gpt-5-image-mini": { "id": "openai/gpt-5-image-mini", "name": "GPT-5 Image Mini", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-10-16", "last_updated": "2025-10-16", "modalities": { "input": ["pdf", "image", "text"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 2.5, "output": 2, "cache_read": 0.25 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Codex GPT for repository edits, code review, and practical software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "openai/gpt-5.1-codex-max": { "id": "openai/gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-oss-120b:free": { "id": "openai/gpt-oss-120b:free", "name": "gpt-oss-120b (free)", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-4o-2024-08-06": { "id": "openai/gpt-4o-2024-08-06", "name": "GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-08-06", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-audio-mini": { "id": "openai/gpt-audio-mini", "name": "GPT Audio Mini", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "o-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.6, "output": 2.4 } }, "openai/gpt-5.6-luna": { "id": "openai/gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 100000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-5.6-terra": { "id": "openai/gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "cache_write": 3.125 } }, "openai/o4-mini-deep-research": { "id": "openai/o4-mini-deep-research", "name": "o4-mini-deep-research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-05", "release_date": "2024-06-26", "last_updated": "2024-06-26", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.036, "output": 0.18 } }, "openai/gpt-4o-2024-11-20": { "id": "openai/gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "openai/o1": { "id": "openai/o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "openai/o1-pro": { "id": "openai/o1-pro", "name": "o1-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2025-03-19", "last_updated": "2025-03-19", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 150, "output": 600 } }, "openai/gpt-chat-latest": { "id": "openai/gpt-chat-latest", "name": "GPT Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-05-05", "last_updated": "2026-05-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "openai/gpt-5-image": { "id": "openai/gpt-5-image", "name": "GPT-5 Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-10-01", "release_date": "2025-10-14", "last_updated": "2025-10-14", "modalities": { "input": ["image", "text", "pdf"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 10, "output": 10, "cache_read": 1.25 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/o3-deep-research": { "id": "openai/o3-deep-research", "name": "o3-deep-research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-05", "release_date": "2024-06-26", "last_updated": "2024-06-26", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 10, "output": 40, "cache_read": 2.5 } }, "openai/gpt-4-turbo-preview": { "id": "openai/gpt-4-turbo-preview", "name": "GPT-4 Turbo Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "openai/gpt-4-turbo": { "id": "openai/gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.01 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "openai/o3-mini-high": { "id": "openai/o3-mini-high", "name": "o3 Mini High", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-10-31", "release_date": "2025-02-12", "last_updated": "2025-02-12", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "openai/gpt-5.4-image-2": { "id": "openai/gpt-5.4-image-2", "name": "GPT-5.4 Image 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": false, "structured_output": true, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["image", "text", "pdf"], "output": ["image", "text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 8, "output": 15, "cache_read": 2 } }, "openai/gpt-4o-search-preview": { "id": "openai/gpt-4o-search-preview", "name": "GPT-4o Search Preview", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": false, "knowledge": "2023-10-31", "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.029, "output": 0.14 } }, "openai/gpt-5.6-sol": { "id": "openai/gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25 } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-oss-20b:free": { "id": "openai/gpt-oss-20b:free", "name": "gpt-oss-20b (free)", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "thedrummer/cydonia-24b-v4.1": { "id": "thedrummer/cydonia-24b-v4.1", "name": "Cydonia 24B V4.1", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2025-09-27", "last_updated": "2025-09-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.3, "output": 0.5, "cache_read": 0.15 } }, "thedrummer/skyfall-36b-v2": { "id": "thedrummer/skyfall-36b-v2", "name": "Skyfall 36B V2", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-03-10", "last_updated": "2025-03-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.55, "output": 0.8, "cache_read": 0.25 } }, "thedrummer/unslopnemo-12b": { "id": "thedrummer/unslopnemo-12b", "name": "UnslopNemo 12B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-11-08", "last_updated": "2024-11-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.4, "output": 0.4 } }, "thedrummer/rocinante-12b": { "id": "thedrummer/rocinante-12b", "name": "Rocinante 12B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-09-30", "last_updated": "2024-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.25, "output": 0.5 } }, "bytedance/ui-tars-1.5-7b": { "id": "bytedance/ui-tars-1.5-7b", "name": "UI-TARS 7B ", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 2048 }, "cost": { "input": 0.1, "output": 0.2, "cache_read": 0.1 } }, "rekaai/reka-flash-3": { "id": "rekaai/reka-flash-3", "name": "Reka Flash 3", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "reka", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.1, "output": 0.2 } }, "rekaai/reka-edge": { "id": "rekaai/reka-edge", "name": "Reka Edge", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "reka", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.1, "output": 0.1 } }, "mistralai/mistral-large-2407": { "id": "mistralai/mistral-large-2407", "name": "Mistral Large 2407", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-03-31", "release_date": "2024-11-19", "last_updated": "2024-11-19", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2 } }, "mistralai/mistral-small-3.2-24b-instruct": { "id": "mistralai/mistral-small-3.2-24b-instruct", "name": "Mistral Small 3.2 24B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-10-31", "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.075, "output": 0.2 } }, "mistralai/mistral-nemo": { "id": "mistralai/mistral-nemo", "name": "Mistral Nemo", "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.02, "output": 0.03 } }, "mistralai/mistral-medium-3-5": { "id": "mistralai/mistral-medium-3-5", "name": "Mistral Medium 3.5", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.5, "output": 7.5 } }, "mistralai/ministral-8b-2512": { "id": "mistralai/ministral-8b-2512", "name": "Ministral 3 8B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.15, "output": 0.15, "cache_read": 0.015 } }, "mistralai/mistral-small-3.1-24b-instruct": { "id": "mistralai/mistral-small-3.1-24b-instruct", "name": "Mistral Small 3.1 24B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-10-31", "release_date": "2025-03-17", "last_updated": "2025-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.351, "output": 0.555 } }, "mistralai/mistral-saba": { "id": "mistralai/mistral-saba", "name": "Saba", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-02-17", "last_updated": "2025-02-17", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.2, "output": 0.6, "cache_read": 0.02 } }, "mistralai/mistral-large": { "id": "mistralai/mistral-large", "name": "Mistral Large", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-11-30", "release_date": "2024-02-26", "last_updated": "2024-02-26", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2 } }, "mistralai/mistral-medium-3.1": { "id": "mistralai/mistral-medium-3.1", "name": "Mistral Medium 3.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-08-13", "last_updated": "2025-08-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 262144 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.04 } }, "mistralai/mistral-small-24b-instruct-2501": { "id": "mistralai/mistral-small-24b-instruct-2501", "name": "Mistral Small 3", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-10-31", "release_date": "2025-01-30", "last_updated": "2025-01-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.05, "output": 0.08 } }, "mistralai/ministral-3b-2512": { "id": "mistralai/ministral-3b-2512", "name": "Ministral 3 3B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.1, "output": 0.1, "cache_read": 0.01 } }, "mistralai/mistral-small-2603": { "id": "mistralai/mistral-small-2603", "name": "Mistral Small 4", "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015 } }, "mistralai/ministral-14b-2512": { "id": "mistralai/ministral-14b-2512", "name": "Ministral 3 14B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2, "output": 0.2, "cache_read": 0.02 } }, "mistralai/devstral-2512": { "id": "mistralai/devstral-2512", "name": "Devstral 2", "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", "family": "devstral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "status": "deprecated", "cost": { "input": 0.4, "output": 2, "cache_read": 0.04 } }, "mistralai/mixtral-8x22b-instruct": { "id": "mistralai/mixtral-8x22b-instruct", "name": "Mixtral 8x22B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-01-31", "release_date": "2024-04-17", "last_updated": "2024-04-17", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 2, "output": 6, "cache_read": 0.2 } }, "mistralai/mistral-medium-3": { "id": "mistralai/mistral-medium-3", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.04 } }, "mistralai/voxtral-small-24b-2507": { "id": "mistralai/voxtral-small-24b-2507", "name": "Voxtral Small 24B 2507", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-30", "last_updated": "2025-10-30", "modalities": { "input": ["text", "audio", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 32000 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.01 } }, "mistralai/mistral-large-2512": { "id": "mistralai/mistral-large-2512", "name": "Mistral Large 3", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.5, "output": 1.5, "cache_read": 0.05 } }, "mistralai/codestral-2508": { "id": "mistralai/codestral-2508", "name": "Codestral 2508", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "codestral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-01", "last_updated": "2025-08-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.3, "output": 0.9, "cache_read": 0.03 } }, "morph/morph-v3-fast": { "id": "morph/morph-v3-fast", "name": "Morph V3 Fast", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-07-07", "last_updated": "2025-07-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 81920, "output": 38000 }, "cost": { "input": 0.8, "output": 1.2 } }, "morph/morph-v3-large": { "id": "morph/morph-v3-large", "name": "Morph V3 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-07-07", "last_updated": "2025-07-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.9, "output": 1.9 } }, "bytedance-seed/seed-1.6-flash": { "id": "bytedance-seed/seed-1.6-flash", "name": "Seed 1.6 Flash", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.075, "output": 0.3 } }, "bytedance-seed/seed-1.6": { "id": "bytedance-seed/seed-1.6", "name": "Seed 1.6", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["image", "text", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.25, "output": 2 } }, "bytedance-seed/seed-2.0-mini": { "id": "bytedance-seed/seed-2.0-mini", "name": "Seed-2.0-Mini", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.1, "output": 0.4 } }, "bytedance-seed/seed-2.0-lite": { "id": "bytedance-seed/seed-2.0-lite", "name": "Seed-2.0-Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-10", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.25, "output": 2 } }, "sakana/fugu-ultra": { "id": "sakana/fugu-ultra", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "family": "fugu", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["max", "xhigh", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-24", "last_updated": "2026-06-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "anthracite-org/magnum-v4-72b": { "id": "anthracite-org/magnum-v4-72b", "name": "Magnum v4 72B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 3, "output": 5 } }, "nvidia/nemotron-3-nano-30b-a3b:free": { "id": "nvidia/nemotron-3-nano-30b-a3b:free", "name": "Nemotron 3 Nano 30B A3B (free)", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-nano-9b-v2:free": { "id": "nvidia/nemotron-nano-9b-v2:free", "name": "Nemotron Nano 9B V2 (free)", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-18", "last_updated": "2025-08-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-nano-12b-v2-vl:free": { "id": "nvidia/nemotron-nano-12b-v2-vl:free", "name": "Nemotron Nano 12B 2 VL (free)", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free": { "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", "name": "Nemotron 3 Nano Omni (free)", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-ultra-550b-a55b:free": { "id": "nvidia/nemotron-3-ultra-550b-a55b:free", "name": "Nemotron 3 Ultra (free)", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-ultra-550b-a55b": { "id": "nvidia/nemotron-3-ultra-550b-a55b", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.5, "output": 2.2, "cache_read": 0.1 } }, "nvidia/nemotron-3.5-content-safety:free": { "id": "nvidia/nemotron-3.5-content-safety:free", "name": "Nemotron 3.5 Content Safety (free)", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-nano-30b-a3b": { "id": "nvidia/nemotron-3-nano-30b-a3b", "name": "Nemotron 3 Nano 30B A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 228000 }, "cost": { "input": 0.05, "output": 0.2 } }, "nvidia/llama-3.3-nemotron-super-49b-v1.5": { "id": "nvidia/llama-3.3-nemotron-super-49b-v1.5", "name": "Llama 3.3 Nemotron Super 49B v1.5", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.4, "output": 0.4 } }, "nvidia/nemotron-3-super-120b-a12b:free": { "id": "nvidia/nemotron-3-super-120b-a12b:free", "name": "Nemotron 3 Super (free)", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 262144 }, "cost": { "input": 0, "output": 0 } }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nemotron 3 Super 120B A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium"] }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.08, "output": 0.45 } }, "cognitivecomputations/dolphin-mistral-24b-venice-edition": { "id": "cognitivecomputations/dolphin-mistral-24b-venice-edition", "name": "Uncensored", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-04-30", "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.2, "output": 0.9 } }, "cognitivecomputations/dolphin-mistral-24b-venice-edition:free": { "id": "cognitivecomputations/dolphin-mistral-24b-venice-edition:free", "name": "Uncensored (free)", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0, "output": 0 } }, "xiaomi/mimo-v2.5": { "id": "xiaomi/mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.105, "output": 0.28, "cache_read": 0.028 } }, "xiaomi/mimo-v2.5-pro": { "id": "xiaomi/mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036 } }, "inception/mercury-2": { "id": "inception/mercury-2", "name": "Mercury 2", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "mercury", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-04", "last_updated": "2026-03-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 50000 }, "cost": { "input": 0.25, "output": 0.75, "cache_read": 0.025 } }, "anthropic/claude-sonnet-4.5": { "id": "anthropic/claude-sonnet-4.5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "anthropic/claude-haiku-4.5": { "id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "anthropic/claude-opus-4.7-fast": { "id": "anthropic/claude-opus-4.7-fast", "name": "Claude Opus 4.7 (Fast)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 } }, "anthropic/claude-sonnet-5": { "id": "anthropic/claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-fable-5": { "id": "anthropic/claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "anthropic/claude-opus-4.1": { "id": "anthropic/claude-opus-4.1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.5": { "id": "anthropic/claude-opus-4.5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "anthropic/claude-3-haiku": { "id": "anthropic/claude-3-haiku", "name": "Claude 3 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-03-13", "last_updated": "2024-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 4096 }, "cost": { "input": 0.25, "output": 1.25, "cache_read": 0.03, "cache_write": 0.3 } }, "anthropic/claude-opus-4": { "id": "anthropic/claude-opus-4", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["image", "text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-opus-4.8-fast": { "id": "anthropic/claude-opus-4.8-fast", "name": "Claude Opus 4.8 (Fast)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-05-27", "last_updated": "2026-05-27", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "tencent/hy3:free": { "id": "tencent/hy3:free", "name": "Hy3 (free)", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hy3", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0 } }, "tencent/hy3": { "id": "tencent/hy3", "name": "Hy3", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hy3", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.14, "output": 0.58, "cache_read": 0.035 } }, "tencent/hunyuan-a13b-instruct": { "id": "tencent/hunyuan-a13b-instruct", "name": "Hunyuan A13B Instruct", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-07-08", "last_updated": "2025-07-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.14, "output": 0.57 } }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Hy3 preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "high"] } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.063, "output": 0.21, "cache_read": 0.021 } }, "deepcogito/cogito-v2.1-671b": { "id": "deepcogito/cogito-v2.1-671b", "name": "Cogito v2.1 671B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "cogito", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 1.25, "output": 1.25 } }, "cohere/command-a": { "id": "cohere/command-a", "name": "Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 2.5, "output": 10 } }, "cohere/command-r-08-2024": { "id": "cohere/command-r-08-2024", "name": "Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.15, "output": 0.6 } }, "cohere/command-r7b-12-2024": { "id": "cohere/command-r7b-12-2024", "name": "Command R7B", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-12-02", "last_updated": "2024-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.0375, "output": 0.15 } }, "cohere/command-r-plus-08-2024": { "id": "cohere/command-r-plus-08-2024", "name": "Command R+", "description": "Cohere's RAG workhorse for long-context enterprise search and tool use", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 2.5, "output": 10 } }, "cohere/north-mini-code:free": { "id": "cohere/north-mini-code:free", "name": "North Mini Code (free)", "description": "Cohere coding model for practical software engineering and agentic edits", "family": "north", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "~x-ai/grok-latest": { "id": "~x-ai/grok-latest", "name": "Grok Latest", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 500000, "output": 1000000 }, "cost": { "input": 2, "output": 6, "cache_read": 0.5 } }, "gryphe/mythomax-l2-13b": { "id": "gryphe/mythomax-l2-13b", "name": "MythoMax 13B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-06-30", "release_date": "2023-07-02", "last_updated": "2023-07-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 0.06, "output": 0.06 } }, "stepfun/step-3.7-flash": { "id": "stepfun/step-3.7-flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 1.15, "cache_read": 0.04 } }, "stepfun/step-3.5-flash": { "id": "stepfun/step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.1, "output": 0.3 } }, "nex-agi/nex-n2-mini": { "id": "nex-agi/nex-n2-mini", "name": "Nex-N2-Mini", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "agi", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-24", "last_updated": "2026-06-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.025, "output": 0.1, "cache_read": 0.0025 } }, "nex-agi/nex-n2-pro": { "id": "nex-agi/nex-n2-pro", "name": "Nex-N2-Pro", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "agi", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-06-08", "last_updated": "2026-06-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.25, "output": 1, "cache_read": 0.025 } }, "undi95/remm-slerp-l2-13b": { "id": "undi95/remm-slerp-l2-13b", "name": "ReMM SLERP 13B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-06-30", "release_date": "2023-07-22", "last_updated": "2023-07-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 6144, "output": 4096 }, "cost": { "input": 0.45, "output": 0.65 } }, "~openai/gpt-mini-latest": { "id": "~openai/gpt-mini-latest", "name": "OpenAI GPT Mini Latest", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "~openai/gpt-latest": { "id": "~openai/gpt-latest", "name": "OpenAI GPT Latest", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["pdf", "image", "text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25 } }, "~moonshotai/kimi-latest": { "id": "~moonshotai/kimi-latest", "name": "MoonshotAI Kimi Latest", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.66, "output": 3.41, "cache_read": 0.15 } }, "relace/relace-search": { "id": "relace/relace-search", "name": "Relace Search", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 1, "output": 3 } }, "relace/relace-apply-3": { "id": "relace/relace-apply-3", "name": "Relace Apply 3", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2025-09-26", "last_updated": "2025-09-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.85, "output": 1.25 } }, "ai21/jamba-large-1.7": { "id": "ai21/jamba-large-1.7", "name": "Jamba Large 1.7", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "jamba", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 4096 }, "cost": { "input": 2, "output": 8 } }, "arcee-ai/coder-large": { "id": "arcee-ai/coder-large", "name": "Coder Large", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-05", "last_updated": "2025-05-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.5, "output": 0.8 } }, "arcee-ai/virtuoso-large": { "id": "arcee-ai/virtuoso-large", "name": "Virtuoso Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-05", "last_updated": "2025-05-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 64000 }, "cost": { "input": 0.75, "output": 1.2 } }, "arcee-ai/trinity-large-thinking": { "id": "arcee-ai/trinity-large-thinking", "name": "Trinity Large Thinking", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "trinity", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 80000 }, "cost": { "input": 0.25, "output": 0.8, "cache_read": 0.06 } }, "mancer/weaver": { "id": "mancer/weaver", "name": "Weaver (alpha)", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "alpha", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-06-30", "release_date": "2023-08-02", "last_updated": "2023-08-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "output": 2000 }, "cost": { "input": 0.75, "output": 1 } }, "perplexity/sonar-reasoning-pro": { "id": "perplexity/sonar-reasoning-pro", "name": "Sonar Reasoning Pro", "description": "Web-grounded reasoning model for multi-step research and cited answers", "family": "sonar-reasoning", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-03-07", "last_updated": "2025-03-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 2, "output": 8 } }, "perplexity/sonar": { "id": "perplexity/sonar", "name": "Sonar", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127072, "output": 127072 }, "cost": { "input": 1, "output": 1 } }, "perplexity/sonar-pro": { "id": "perplexity/sonar-pro", "name": "Sonar Pro", "description": "Advanced Sonar search model for deeper research and cited synthesis", "family": "sonar-pro", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-03-07", "last_updated": "2025-03-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8000 }, "cost": { "input": 3, "output": 15 } }, "perplexity/sonar-pro-search": { "id": "perplexity/sonar-pro-search", "name": "Sonar Pro Search", "description": "Advanced Sonar search model for deeper research and cited synthesis", "family": "sonar-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-10-30", "last_updated": "2025-10-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8000 }, "cost": { "input": 3, "output": 15 } }, "perplexity/sonar-deep-research": { "id": "perplexity/sonar-deep-research", "name": "Sonar Deep Research", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar-deep-research", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-03-07", "last_updated": "2025-03-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 2, "output": 8, "reasoning": 3 } }, "openrouter/bodybuilder": { "id": "openrouter/bodybuilder", "name": "Body Builder (beta)", "description": "Preview model for early access evaluation, prototyping, and compatibility testing", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2025-12-05", "last_updated": "2025-12-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 } }, "openrouter/free": { "id": "openrouter/free", "name": "Free Models Router", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 8000 }, "cost": { "input": 0, "output": 0 } }, "openrouter/fusion": { "id": "openrouter/fusion", "name": "Fusion", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 } }, "openrouter/pareto-code": { "id": "openrouter/pareto-code", "name": "Pareto Code Router", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 200000 } }, "openrouter/auto": { "id": "openrouter/auto", "name": "Auto Router", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "auto", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2023-11-08", "last_updated": "2023-11-08", "modalities": { "input": ["text", "image", "audio", "pdf", "video"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 } }, "qwen/qwen3.5-plus-20260420": { "id": "qwen/qwen3.5-plus-20260420", "name": "Qwen3.5 Plus 2026-04-20", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.8, "cache_write": 0.375 } }, "qwen/qwen3-next-80b-a3b-instruct:free": { "id": "qwen/qwen3-next-80b-a3b-instruct:free", "name": "Qwen3 Next 80B A3B Instruct (free)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3-vl-235b-a22b-thinking": { "id": "qwen/qwen3-vl-235b-a22b-thinking", "name": "Qwen3 VL 235B A22B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.26, "output": 2.6 } }, "qwen/qwen3-vl-30b-a3b-thinking": { "id": "qwen/qwen3-vl-30b-a3b-thinking", "name": "Qwen3 VL 30B A3B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.13, "output": 1.56 } }, "qwen/qwen3-coder-plus": { "id": "qwen/qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Hosted Qwen coder for software agents, repo edits, and long-context code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.65, "output": 3.25, "cache_read": 0.13, "cache_write": 0.8125 } }, "qwen/qwen-plus": { "id": "qwen/qwen-plus", "name": "Qwen Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.26, "output": 0.78, "cache_read": 0.052, "cache_write": 0.325 } }, "qwen/qwen3-coder-30b-a3b-instruct": { "id": "qwen/qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 160000, "output": 32768 }, "cost": { "input": 0.07, "output": 0.27 } }, "qwen/qwen3-32b": { "id": "qwen/qwen3-32b", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.08, "output": 0.28 } }, "qwen/qwen3-next-80b-a3b-instruct": { "id": "qwen/qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.09, "output": 1.1 } }, "qwen/qwen3-vl-8b-instruct": { "id": "qwen/qwen3-vl-8b-instruct", "name": "Qwen3 VL 8B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-14", "last_updated": "2025-10-14", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.117, "output": 0.455 } }, "qwen/qwen3.7-plus": { "id": "qwen/qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 262144 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.32, "output": 1.28, "cache_read": 0.064, "cache_write": 0.4 } }, "qwen/qwen3.6-35b-a3b": { "id": "qwen/qwen3.6-35b-a3b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.14, "output": 1 } }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 262144 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 1.25, "output": 3.75, "cache_read": 0.25, "cache_write": 1.5625 } }, "qwen/qwen3-max": { "id": "qwen/qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.78, "output": 3.9, "cache_read": 0.156, "cache_write": 0.975 } }, "qwen/qwen3-8b": { "id": "qwen/qwen3-8b", "name": "Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.117, "output": 0.455 } }, "qwen/qwen-plus-2025-07-28": { "id": "qwen/qwen-plus-2025-07-28", "name": "Qwen Plus 0728", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.26, "output": 0.78 } }, "qwen/qwen3.5-flash-02-23": { "id": "qwen/qwen3.5-flash-02-23", "name": "Qwen3.5-Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-25", "last_updated": "2026-02-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.065, "output": 0.26 } }, "qwen/qwen3-coder:free": { "id": "qwen/qwen3-coder:free", "name": "Qwen3 Coder 480B A35B (free)", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 262000 }, "cost": { "input": 0, "output": 0 } }, "qwen/qwen3-30b-a3b-instruct-2507": { "id": "qwen/qwen3-30b-a3b-instruct-2507", "name": "Qwen3 30B A3B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32000 }, "cost": { "input": 0.04815, "output": 0.19305 } }, "qwen/qwen-2.5-coder-32b-instruct": { "id": "qwen/qwen-2.5-coder-32b-instruct", "name": "Qwen2.5 Coder 32B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-11-11", "last_updated": "2024-11-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.66, "output": 1 } }, "qwen/qwen3-next-80b-a3b-thinking": { "id": "qwen/qwen3-next-80b-a3b-thinking", "name": "Qwen3-Next 80B-A3B (Thinking)", "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.0975, "output": 0.78 } }, "qwen/qwen3-235b-a22b-thinking-2507": { "id": "qwen/qwen3-235b-a22b-thinking-2507", "name": "Qwen3 235B A22B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.1495, "output": 1.495 } }, "qwen/qwen3-vl-32b-instruct": { "id": "qwen/qwen3-vl-32b-instruct", "name": "Qwen3 VL 32B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-23", "last_updated": "2025-10-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.104, "output": 0.416 } }, "qwen/qwen3-coder": { "id": "qwen/qwen3-coder", "name": "Qwen3 Coder 480B A35B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.22, "output": 1.8 } }, "qwen/qwen3.6-flash": { "id": "qwen/qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1875, "output": 1.125, "cache_write": 0.234375 } }, "qwen/qwen3.5-plus-02-15": { "id": "qwen/qwen3.5-plus-02-15", "name": "Qwen3.5 Plus 2026-02-15", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.26, "output": 1.56 } }, "qwen/qwen-2.5-7b-instruct": { "id": "qwen/qwen-2.5-7b-instruct", "name": "Qwen2.5 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-10-16", "last_updated": "2024-10-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.04, "output": 0.1 } }, "qwen/qwen3-vl-8b-thinking": { "id": "qwen/qwen3-vl-8b-thinking", "name": "Qwen3 VL 8B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-14", "last_updated": "2025-10-14", "modalities": { "input": ["image", "text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 32768 }, "cost": { "input": 0.117, "output": 1.365 } }, "qwen/qwen3-max-thinking": { "id": "qwen/qwen3-max-thinking", "name": "Qwen3 Max Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-09", "last_updated": "2026-02-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.78, "output": 3.9 } }, "qwen/qwen3-30b-a3b-thinking-2507": { "id": "qwen/qwen3-30b-a3b-thinking-2507", "name": "Qwen3 30B A3B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.13, "output": 1.56 } }, "qwen/qwen2.5-vl-72b-instruct": { "id": "qwen/qwen2.5-vl-72b-instruct", "name": "Qwen2.5 VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-02-01", "last_updated": "2025-02-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 128000 }, "cost": { "input": 0.25, "output": 0.75 } }, "qwen/qwen3.5-27b": { "id": "qwen/qwen3.5-27b", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.195, "output": 1.56 } }, "qwen/qwen3-235b-a22b": { "id": "qwen/qwen3-235b-a22b", "name": "Qwen3 235B-A22B", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 38912 } ], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.455, "output": 1.82 } }, "qwen/qwen-2.5-72b-instruct": { "id": "qwen/qwen-2.5-72b-instruct", "name": "Qwen2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.36, "output": 0.4 } }, "qwen/qwen-plus-2025-07-28:thinking": { "id": "qwen/qwen-plus-2025-07-28:thinking", "name": "Qwen Plus 0728 (thinking)", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.26, "output": 0.78, "cache_write": 0.325 } }, "qwen/qwen3-coder-next": { "id": "qwen/qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-04", "last_updated": "2026-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.11, "output": 0.8, "cache_read": 0.07 } }, "qwen/qwen3.6-27b": { "id": "qwen/qwen3.6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262140 }, "cost": { "input": 0.285, "output": 2.4, "cache_read": 0.15 } }, "qwen/qwen3.5-35b-a3b": { "id": "qwen/qwen3.5-35b-a3b", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 81920 }, "cost": { "input": 0.14, "output": 1, "cache_read": 0.05 } }, "qwen/qwen3.5-9b": { "id": "qwen/qwen3.5-9b", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.1, "output": 0.15 } }, "qwen/qwen3.5-397b-a17b": { "id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.385, "output": 2.45, "cache_read": 0.111 } }, "qwen/qwen3-vl-30b-a3b-instruct": { "id": "qwen/qwen3-vl-30b-a3b-instruct", "name": "Qwen3 VL 30B A3B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.13, "output": 0.52 } }, "qwen/qwen3-235b-a22b-2507": { "id": "qwen/qwen3-235b-a22b-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-21", "last_updated": "2025-07-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.09, "output": 0.55 } }, "qwen/qwen3-coder-flash": { "id": "qwen/qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.195, "output": 0.975, "cache_read": 0.039, "cache_write": 0.24375 } }, "qwen/qwen3-14b": { "id": "qwen/qwen3-14b", "name": "Qwen3 14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131702, "output": 40960 }, "cost": { "input": 0.1, "output": 0.24 } }, "qwen/qwen3-vl-235b-a22b-instruct": { "id": "qwen/qwen3-vl-235b-a22b-instruct", "name": "Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0.2, "output": 0.88, "cache_read": 0.11 } }, "qwen/qwen3-30b-a3b": { "id": "qwen/qwen3-30b-a3b", "name": "Qwen3 30B A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.12, "output": 0.5 } }, "qwen/qwen3.6-max-preview": { "id": "qwen/qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 131072 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.04, "output": 6.24, "cache_write": 1.3 } }, "qwen/qwen3.5-122b-a10b": { "id": "qwen/qwen3.5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.26, "output": 2.08 } }, "qwen/qwen3.6-plus": { "id": "qwen/qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1, "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.325, "output": 1.95, "cache_write": 0.40625 } }, "amazon/nova-lite-v1": { "id": "amazon/nova-lite-v1", "name": "Nova Lite 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-lite", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 5120 }, "cost": { "input": 0.06, "output": 0.24 } }, "amazon/nova-premier-v1": { "id": "amazon/nova-premier-v1", "name": "Nova Premier 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-31", "last_updated": "2025-10-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 2.5, "output": 12.5, "cache_read": 0.625 } }, "amazon/nova-pro-v1": { "id": "amazon/nova-pro-v1", "name": "Nova Pro 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova-pro", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "output": 5120 }, "cost": { "input": 0.8, "output": 3.2 } }, "amazon/nova-micro-v1": { "id": "amazon/nova-micro-v1", "name": "Nova Micro 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-micro", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 5120 }, "cost": { "input": 0.035, "output": 0.14 } }, "amazon/nova-2-lite-v1": { "id": "amazon/nova-2-lite-v1", "name": "Nova 2 Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "nova", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65535 }, "cost": { "input": 0.3, "output": 2.5 } }, "aion-labs/aion-3.0-mini": { "id": "aion-labs/aion-3.0-mini", "name": "Aion-3.0-Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-07", "last_updated": "2026-07-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.7, "output": 1.4, "cache_read": 0.18 } }, "aion-labs/aion-rp-llama-3.1-8b": { "id": "aion-labs/aion-rp-llama-3.1-8b", "name": "Aion-RP 1.0 (8B)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-12-31", "release_date": "2025-02-04", "last_updated": "2025-02-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.8, "output": 1.6 } }, "aion-labs/aion-3.0": { "id": "aion-labs/aion-3.0", "name": "Aion-3.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-07", "last_updated": "2026-07-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 3, "output": 6, "cache_read": 0.75 } }, "aion-labs/aion-2.0": { "id": "aion-labs/aion-2.0", "name": "Aion-2.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.8, "output": 1.6, "cache_read": 0.2 } }, "inflection/inflection-3-pi": { "id": "inflection/inflection-3-pi", "name": "Inflection 3 Pi", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-10-11", "last_updated": "2024-10-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "output": 1024 }, "cost": { "input": 2.5, "output": 10 } }, "inflection/inflection-3-productivity": { "id": "inflection/inflection-3-productivity", "name": "Inflection 3 Productivity", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-10-11", "last_updated": "2024-10-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "output": 1024 }, "cost": { "input": 2.5, "output": 10 } }, "sao10k/l3.1-euryale-70b": { "id": "sao10k/l3.1-euryale-70b", "name": "Llama 3.1 Euryale 70B v2.2", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-28", "last_updated": "2024-08-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.85, "output": 0.85 } }, "sao10k/l3.3-euryale-70b": { "id": "sao10k/l3.3-euryale-70b", "name": "Llama 3.3 Euryale 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-12-18", "last_updated": "2024-12-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.65, "output": 0.75 } }, "sao10k/l3-lunaris-8b": { "id": "sao10k/l3-lunaris-8b", "name": "Llama 3 8B Lunaris", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-13", "last_updated": "2024-08-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 16384 }, "cost": { "input": 0.04, "output": 0.05 } }, "sao10k/l3.1-70b-hanami-x1": { "id": "sao10k/l3.1-70b-hanami-x1", "name": "Llama 3.1 70B Hanami x1", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2025-01-08", "last_updated": "2025-01-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 16000, "output": 16000 }, "cost": { "input": 3, "output": 3 } }, "upstage/solar-pro-3": { "id": "upstage/solar-pro-3", "name": "Solar Pro 3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "solar-pro", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015 } }, "allenai/olmo-3-32b-think": { "id": "allenai/olmo-3-32b-think", "name": "Olmo 3 32B Think", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "allenai", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-11-21", "last_updated": "2025-11-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.15, "output": 0.5 } }, "deepseek/deepseek-r1-0528": { "id": "deepseek/deepseek-r1-0528", "name": "R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.5, "output": 2.15, "cache_read": 0.35 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 0.077, "output": 0.154, "cache_read": 0.0154 } }, "deepseek/deepseek-v3.1-terminus": { "id": "deepseek/deepseek-v3.1-terminus", "name": "DeepSeek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.27, "output": 0.95, "cache_read": 0.13 } }, "deepseek/deepseek-r1-distill-llama-70b": { "id": "deepseek/deepseek-r1-distill-llama-70b", "name": "R1 Distill Llama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-07-31", "release_date": "2025-01-23", "last_updated": "2025-01-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.8, "output": 0.8 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "deepseek/deepseek-r1": { "id": "deepseek/deepseek-r1", "name": "DeepSeek-R1", "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 16000 }, "cost": { "input": 0.7, "output": 2.5 } }, "deepseek/deepseek-v3.2-exp": { "id": "deepseek/deepseek-v3.2-exp", "name": "DeepSeek V3.2 Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "cost": { "input": 0.27, "output": 0.41 } }, "deepseek/deepseek-chat-v3-0324": { "id": "deepseek/deepseek-chat-v3-0324", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 16384 }, "cost": { "input": 0.24, "output": 0.9, "cache_read": 0.135 } }, "deepseek/deepseek-chat": { "id": "deepseek/deepseek-chat", "name": "DeepSeek Chat", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16000 }, "cost": { "input": 0.2002, "output": 0.8001 } }, "deepseek/deepseek-v3.2": { "id": "deepseek/deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 64000 }, "cost": { "input": 0.2145, "output": 0.32175, "cache_read": 0.02145 } }, "deepseek/deepseek-chat-v3.1": { "id": "deepseek/deepseek-chat-v3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.21, "output": 0.79, "cache_read": 0.13 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 196608 }, "cost": { "input": 0.15, "output": 0.9, "cache_read": 0.05 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "MiniMax-M2.1", "description": "Earlier MiniMax agent model for practical coding and productivity tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "minimax/minimax-01": { "id": "minimax/minimax-01", "name": "MiniMax-01", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-03-31", "release_date": "2025-01-15", "last_updated": "2025-01-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000192, "output": 1000192 }, "cost": { "input": 0.2, "output": 1.1 } }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_details" }, "structured_output": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.255, "output": 1.02 } }, "minimax/minimax-m2-her": { "id": "minimax/minimax-m2-her", "name": "MiniMax M2-her", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01-23", "last_updated": "2026-01-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "output": 2048 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 196608 }, "cost": { "input": 0.24, "output": 0.96 } }, "minimax/minimax-m1": { "id": "minimax/minimax-m1", "name": "MiniMax M1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 40000 }, "cost": { "input": 0.4, "output": 2.2 } }, "kwaipilot/kat-coder-pro-v2": { "id": "kwaipilot/kat-coder-pro-v2", "name": "KAT-Coder-Pro V2", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 80000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "nousresearch/hermes-3-llama-3.1-405b:free": { "id": "nousresearch/hermes-3-llama-3.1-405b:free", "name": "Hermes 3 405B Instruct (free)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "hermes", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-16", "last_updated": "2024-08-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "nousresearch/hermes-4-405b": { "id": "nousresearch/hermes-4-405b", "name": "Hermes 4 405B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "hermes", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 1, "output": 3 } }, "nousresearch/hermes-3-llama-3.1-70b": { "id": "nousresearch/hermes-3-llama-3.1-70b", "name": "Hermes 3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "nousresearch", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-18", "last_updated": "2024-08-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.7, "output": 0.7 } }, "nousresearch/hermes-3-llama-3.1-405b": { "id": "nousresearch/hermes-3-llama-3.1-405b", "name": "Hermes 3 405B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "nousresearch", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-16", "last_updated": "2024-08-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 1, "output": 1 } }, "nousresearch/hermes-4-70b": { "id": "nousresearch/hermes-4-70b", "name": "Hermes 4 70B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "hermes", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.13, "output": 0.4 } } } }, "jiekou": { "id": "jiekou", "env": ["JIEKOU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.jiekou.ai/openai", "name": "Jiekou.AI", "doc": "https://docs.jiekou.ai/docs/support/quickstart?utm_source=github_models.dev", "models": { "o3": { "id": "o3", "name": "o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 10, "output": 40 } }, "grok-code-fast-1": { "id": "grok-code-fast-1", "name": "grok-code-fast-1", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.18, "output": 1.35 } }, "gpt-5.2-pro": { "id": "gpt-5.2-pro", "name": "gpt-5.2-pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 18.9, "output": 151.2 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "gemini-2.5-pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 1.125, "output": 9 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "claude-haiku-4-5-20251001", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 20000, "output": 64000 }, "cost": { "input": 0.9, "output": 4.5 } }, "gpt-5-pro": { "id": "gpt-5-pro", "name": "gpt-5-pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 272000 }, "cost": { "input": 13.5, "output": 108 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "gemini-2.5-flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.27, "output": 2.25 } }, "o4-mini": { "id": "o4-mini", "name": "o4-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4 } }, "gemini-2.5-flash-lite-preview-09-2025": { "id": "gemini-2.5-flash-lite-preview-09-2025", "name": "gemini-2.5-flash-lite-preview-09-2025", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.09, "output": 0.36 } }, "claude-opus-4-1-20250805": { "id": "claude-opus-4-1-20250805", "name": "claude-opus-4-1-20250805", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 13.5, "output": 67.5 } }, "grok-4-1-fast-non-reasoning": { "id": "grok-4-1-fast-non-reasoning", "name": "grok-4-1-fast-non-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.18, "output": 0.45 } }, "gpt-5-chat-latest": { "id": "gpt-5-chat-latest", "name": "gpt-5-chat-latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.125, "output": 9 } }, "claude-opus-4-5-20251101": { "id": "claude-opus-4-5-20251101", "name": "claude-opus-4-5-20251101", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 65536 }, "cost": { "input": 4.5, "output": 22.5 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "gpt-5.1-codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.125, "output": 9 } }, "gpt-5.1-codex-max": { "id": "gpt-5.1-codex-max", "name": "gpt-5.1-codex-max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.125, "output": 9 } }, "claude-opus-4-20250514": { "id": "claude-opus-4-20250514", "name": "claude-opus-4-20250514", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 13.5, "output": 67.5 } }, "o3-mini": { "id": "o3-mini", "name": "o3-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 1.1, "output": 4.4 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "gpt-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.575, "output": 12.6 } }, "grok-4-0709": { "id": "grok-4-0709", "name": "grok-4-0709", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 8192 }, "cost": { "input": 2.7, "output": 13.5 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "gemini-2.5-flash-lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.09, "output": 0.36 } }, "claude-sonnet-4-20250514": { "id": "claude-sonnet-4-20250514", "name": "claude-sonnet-4-20250514", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 2.7, "output": 13.5 } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "gpt-5.1-codex-mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.225, "output": 1.8 } }, "grok-4-fast-reasoning": { "id": "grok-4-fast-reasoning", "name": "grok-4-fast-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.18, "output": 0.45 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "claude-opus-4-6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02", "last_updated": "2026-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "claude-sonnet-4-5-20250929", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 2.7, "output": 13.5 } }, "grok-4-1-fast-reasoning": { "id": "grok-4-1-fast-reasoning", "name": "grok-4-1-fast-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.18, "output": 0.45 } }, "gemini-3-pro-preview": { "id": "gemini-3-pro-preview", "name": "gemini-3-pro-preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.8, "output": 10.8 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "gpt-5-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.225, "output": 1.8 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "gpt-5-nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.045, "output": 0.36 } }, "grok-4-fast-non-reasoning": { "id": "grok-4-fast-non-reasoning", "name": "grok-4-fast-non-reasoning", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 2000000 }, "cost": { "input": 0.18, "output": 0.45 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "gemini-3-flash-preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3 } }, "gemini-2.5-flash-preview-05-20": { "id": "gemini-2.5-flash-preview-05-20", "name": "gemini-2.5-flash-preview-05-20", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 200000 }, "cost": { "input": 0.135, "output": 3.15 } }, "gemini-2.5-flash-lite-preview-06-17": { "id": "gemini-2.5-flash-lite-preview-06-17", "name": "gemini-2.5-flash-lite-preview-06-17", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "video", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 0.09, "output": 0.36 } }, "gemini-2.5-pro-preview-06-05": { "id": "gemini-2.5-pro-preview-06-05", "name": "gemini-2.5-pro-preview-06-05", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 200000 }, "cost": { "input": 1.125, "output": 9 } }, "gpt-5-codex": { "id": "gpt-5-codex", "name": "gpt-5-codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.125, "output": 9 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "gpt-5.2-codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "gpt-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02", "last_updated": "2026-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.125, "output": 9 } }, "moonshotai/kimi-k2-instruct": { "id": "moonshotai/kimi-k2-instruct", "name": "Kimi K2 Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.57, "output": 2.3 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 262143 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 3 } }, "moonshotai/kimi-k2-0905": { "id": "moonshotai/kimi-k2-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5 } }, "minimaxai/minimax-m1-80k": { "id": "minimaxai/minimax-m1-80k", "name": "MiniMax M1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 40000 }, "cost": { "input": 0.55, "output": 2.2 } }, "baidu/ernie-4.5-vl-424b-a47b": { "id": "baidu/ernie-4.5-vl-424b-a47b", "name": "ERNIE 4.5 VL 424B A47B", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "ernie", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 123000, "output": 16000 }, "cost": { "input": 0.42, "output": 1.25 } }, "baidu/ernie-4.5-300b-a47b-paddle": { "id": "baidu/ernie-4.5-300b-a47b-paddle", "name": "ERNIE 4.5 300B A47B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "ernie", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 123000, "output": 12000 }, "cost": { "input": 0.28, "output": 1.1 } }, "xiaomimimo/mimo-v2-flash": { "id": "xiaomimimo/mimo-v2-flash", "name": "XiaomiMiMo/MiMo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "zai-org/glm-4.7": { "id": "zai-org/glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.2 } }, "zai-org/glm-4.5v": { "id": "zai-org/glm-4.5v", "name": "GLM 4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glmv", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.6, "output": 1.8 } }, "zai-org/glm-4.5": { "id": "zai-org/glm-4.5", "name": "GLM-4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0.6, "output": 2.2 } }, "zai-org/glm-4.7-flash": { "id": "zai-org/glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.07, "output": 0.4 } }, "qwen/qwen3-235b-a22b-instruct-2507": { "id": "qwen/qwen3-235b-a22b-instruct-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.15, "output": 0.8 } }, "qwen/qwen3-next-80b-a3b-instruct": { "id": "qwen/qwen3-next-80b-a3b-instruct", "name": "Qwen3 Next 80B A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.15, "output": 1.5 } }, "qwen/qwen3-next-80b-a3b-thinking": { "id": "qwen/qwen3-next-80b-a3b-thinking", "name": "Qwen3 Next 80B A3B Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.15, "output": 1.5 } }, "qwen/qwen3-235b-a22b-thinking-2507": { "id": "qwen/qwen3-235b-a22b-thinking-2507", "name": "Qwen3 235B A22b Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.3, "output": 3 } }, "qwen/qwen3-32b-fp8": { "id": "qwen/qwen3-32b-fp8", "name": "Qwen3 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 20000 }, "cost": { "input": 0.1, "output": 0.45 } }, "qwen/qwen3-30b-a3b-fp8": { "id": "qwen/qwen3-30b-a3b-fp8", "name": "Qwen3 30B A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 20000 }, "cost": { "input": 0.09, "output": 0.45 } }, "qwen/qwen3-coder-next": { "id": "qwen/qwen3-coder-next", "name": "qwen/qwen3-coder-next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02", "last_updated": "2026-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.2, "output": 1.5 } }, "qwen/qwen3-coder-480b-a35b-instruct": { "id": "qwen/qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.29, "output": 1.2 } }, "qwen/qwen3-235b-a22b-fp8": { "id": "qwen/qwen3-235b-a22b-fp8", "name": "Qwen3 235B A22B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 20000 }, "cost": { "input": 0.2, "output": 0.8 } }, "deepseek/deepseek-r1-0528": { "id": "deepseek/deepseek-r1-0528", "name": "DeepSeek R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.7, "output": 2.5 } }, "deepseek/deepseek-v3-0324": { "id": "deepseek/deepseek-v3-0324", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.28, "output": 1.14 } }, "deepseek/deepseek-v3.1": { "id": "deepseek/deepseek-v3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32767 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "cost": { "input": 0.27, "output": 1 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "Minimax M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 131071 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } } } }, "nova": { "id": "nova", "env": ["NOVA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.nova.amazon.com/v1", "name": "Nova", "doc": "https://nova.amazon.com/dev/documentation", "models": { "nova-2-pro-v1": { "id": "nova-2-pro-v1", "name": "Nova 2 Pro", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "nova-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-03", "last_updated": "2026-01-03", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "reasoning": 0 } }, "nova-2-lite-v1": { "id": "nova-2-lite-v1", "name": "Nova 2 Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "nova-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "reasoning": 0 } } } }, "alibaba-token-plan": { "id": "alibaba-token-plan", "env": ["ALIBABA_TOKEN_PLAN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1", "name": "Alibaba Token Plan", "doc": "https://www.alibabacloud.com/help/en/model-studio/token-plan-overview", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "wan2.7-image-pro": { "id": "wan2.7-image-pro", "name": "Wan2.7 Image Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen-image-2.0": { "id": "qwen-image-2.0", "name": "Qwen Image 2.0", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "input": 196601, "output": 24576 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "qwen-image-2.0-pro": { "id": "qwen-image-2.0-pro", "name": "Qwen Image 2.0 Pro", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "wan2.7-image": { "id": "wan2.7-image", "name": "Wan2.7 Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 8192, "output": 0 }, "cost": { "input": 0, "output": 0 } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "output": 16384 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-03", "last_updated": "2025-12-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 81920 } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "alibaba": { "id": "alibaba", "env": ["DASHSCOPE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1", "name": "Alibaba", "doc": "https://www.alibabacloud.com/help/en/model-studio/models", "models": { "qwen3-omni-flash": { "id": "qwen3-omni-flash", "name": "Qwen3-Omni Flash", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.43, "output": 1.66, "input_audio": 3.81, "output_audio": 15.11 } }, "qwen3-coder-plus": { "id": "qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Hosted Qwen coder for software agents, repo edits, and long-context code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1, "output": 5 } }, "qwen-plus": { "id": "qwen-plus", "name": "Qwen Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.4, "output": 1.2, "reasoning": 4 } }, "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.45, "output": 2.25 } }, "qwen3-omni-flash-realtime": { "id": "qwen3-omni-flash-realtime", "name": "Qwen3-Omni Flash Realtime", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 65536, "output": 16384 }, "cost": { "input": 0.52, "output": 1.99, "input_audio": 4.57, "output_audio": 18.13 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.7, "output": 2.8, "reasoning": 8.4 } }, "qwen-omni-turbo-realtime": { "id": "qwen-omni-turbo-realtime", "name": "Qwen-Omni Turbo Realtime", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-05-08", "last_updated": "2025-05-08", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0.27, "output": 1.07, "input_audio": 4.44, "output_audio": 8.89 } }, "qwen-plus-character-ja": { "id": "qwen-plus-character-ja", "name": "Qwen Plus Character (Japanese)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01", "last_updated": "2024-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 512 }, "cost": { "input": 0.5, "output": 1.4 } }, "qwen3-next-80b-a3b-instruct": { "id": "qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 2 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-04", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } }, "qwen3.6-35b-a3b": { "id": "qwen3.6-35b-a3b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.248, "output": 1.485 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.5, "cache_write": 3.125 } }, "qwen3-max": { "id": "qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.2, "output": 6 } }, "qwen2-5-omni-7b": { "id": "qwen2-5-omni-7b", "name": "Qwen2.5-Omni 7B", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-12", "last_updated": "2024-12", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": true, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0.1, "output": 0.4, "input_audio": 6.76 } }, "qwen3-8b": { "id": "qwen3-8b", "name": "Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.18, "output": 0.7, "reasoning": 2.1 } }, "qwen2-5-14b-instruct": { "id": "qwen2-5-14b-instruct", "name": "Qwen2.5 14B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.35, "output": 1.4 } }, "qwen3-next-80b-a3b-thinking": { "id": "qwen3-next-80b-a3b-thinking", "name": "Qwen3-Next 80B-A3B (Thinking)", "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 6 } }, "qvq-max": { "id": "qvq-max", "name": "QVQ Max", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qvq", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 1.2, "output": 4.8 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.1875, "output": 1.125, "cache_write": 0.234375 } }, "qwen2-5-vl-72b-instruct": { "id": "qwen2-5-vl-72b-instruct", "name": "Qwen2.5-VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 2.8, "output": 8.4 } }, "qwen3-vl-plus": { "id": "qwen3-vl-plus", "name": "Qwen3-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.2, "output": 1.6, "reasoning": 4.8 } }, "qwen-vl-ocr": { "id": "qwen-vl-ocr", "name": "Qwen-VL OCR", "description": "OCR model for extracting structured text from documents and screenshots", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2024-10-28", "last_updated": "2025-04-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 34096, "output": 4096 }, "cost": { "input": 0.72, "output": 0.72 } }, "qwen-mt-turbo": { "id": "qwen-mt-turbo", "name": "Qwen-MT Turbo", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 8192 }, "cost": { "input": 0.16, "output": 0.49 } }, "qwen-mt-plus": { "id": "qwen-mt-plus", "name": "Qwen-MT Plus", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01", "last_updated": "2025-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 8192 }, "cost": { "input": 2.46, "output": 7.37 } }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.4, "output": 2.4, "reasoning": 2.4 } }, "qwen-omni-turbo": { "id": "qwen-omni-turbo", "name": "Qwen-Omni Turbo", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01-19", "last_updated": "2025-03-26", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 32768, "output": 2048 }, "cost": { "input": 0.07, "output": 0.27, "input_audio": 4.44, "output_audio": 8.89 } }, "qwen2-5-72b-instruct": { "id": "qwen2-5-72b-instruct", "name": "Qwen2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 1.4, "output": 5.6 } }, "qwen-flash": { "id": "qwen-flash", "name": "Qwen Flash", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32768 }, "cost": { "input": 0.05, "output": 0.4 } }, "qwen3-vl-235b-a22b": { "id": "qwen3-vl-235b-a22b", "name": "Qwen3-VL 235B-A22B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.7, "output": 2.8, "reasoning": 8.4 } }, "qwen3-vl-30b-a3b": { "id": "qwen3-vl-30b-a3b", "name": "Qwen3-VL 30B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.2, "output": 0.8, "reasoning": 2.4 } }, "qwen-vl-max": { "id": "qwen-vl-max", "name": "Qwen-VL Max", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-08", "last_updated": "2025-08-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.8, "output": 3.2 } }, "qwen3.5-27b": { "id": "qwen3.5-27b", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2.4 } }, "qwen-max": { "id": "qwen-max", "name": "Qwen Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-03", "last_updated": "2025-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 1.6, "output": 6.4 } }, "qwen3-235b-a22b": { "id": "qwen3-235b-a22b", "name": "Qwen3 235B-A22B", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.7, "output": 2.8, "reasoning": 8.4 } }, "qwen3-livetranslate-flash-realtime": { "id": "qwen3-livetranslate-flash-realtime", "name": "Qwen3-LiveTranslate Flash Realtime", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 53248, "output": 4096 }, "cost": { "input": 10, "output": 10, "input_audio": 10, "output_audio": 38 } }, "qwen3.6-27b": { "id": "qwen3.6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.6, "output": 3.6 } }, "qwen3.5-35b-a3b": { "id": "qwen3.5-35b-a3b", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.25, "output": 2 } }, "qwen3-coder-480b-a35b-instruct": { "id": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3-Coder 480B-A35B Instruct", "description": "Open Qwen coding heavyweight for repository reasoning and agentic engineering", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.5, "output": 7.5 } }, "qwq-plus": { "id": "qwq-plus", "name": "QwQ Plus", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-03-05", "last_updated": "2025-03-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.8, "output": 2.4 } }, "qwen2-5-32b-instruct": { "id": "qwen2-5-32b-instruct", "name": "Qwen2.5 32B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.7, "output": 2.8 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.6, "output": 3.6 } }, "qwen3-coder-flash": { "id": "qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.5 } }, "qwen3-14b": { "id": "qwen3-14b", "name": "Qwen3 14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.35, "output": 1.4, "reasoning": 4.2 } }, "qwen3-asr-flash": { "id": "qwen3-asr-flash", "name": "Qwen3-ASR Flash", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-04", "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": { "input": ["audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 53248, "output": 4096 }, "cost": { "input": 0.035, "output": 0.035 } }, "qwen-turbo": { "id": "qwen-turbo", "name": "Qwen Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-11-01", "last_updated": "2025-04-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.05, "output": 0.2, "reasoning": 0.5 } }, "qwen2-5-7b-instruct": { "id": "qwen2-5-7b-instruct", "name": "Qwen2.5 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.175, "output": 0.7 } }, "qwen2-5-vl-7b-instruct": { "id": "qwen2-5-vl-7b-instruct", "name": "Qwen2.5-VL 7B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.35, "output": 1.05 } }, "qwen3.6-max-preview": { "id": "qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.3, "output": 7.8, "cache_read": 0.13, "cache_write": 1.625 } }, "qwen3.5-122b-a10b": { "id": "qwen3.5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 3.2 } }, "qwen-vl-plus": { "id": "qwen-vl-plus", "name": "Qwen-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-08-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.21, "output": 0.63 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5 } } } } }, "databricks": { "id": "databricks", "env": ["DATABRICKS_HOST", "DATABRICKS_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://${DATABRICKS_HOST}/ai-gateway/mlflow/v1", "name": "Databricks", "doc": "https://docs.databricks.com/aws/en/machine-learning/foundation-models/", "models": { "databricks-claude-opus-4-7": { "id": "databricks-claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "databricks-gpt-5-4": { "id": "databricks-gpt-5-4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "databricks-gemini-3-flash": { "id": "databricks-gemini-3-flash", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1 } }, "databricks-claude-opus-4-5": { "id": "databricks-claude-opus-4-5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "databricks-gpt-5-nano": { "id": "databricks-gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "databricks-gpt-5-mini": { "id": "databricks-gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "databricks-gpt-5": { "id": "databricks-gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "databricks-gemini-2-5-pro": { "id": "databricks-gemini-2-5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "databricks-gemini-3-1-pro": { "id": "databricks-gemini-3-1-pro", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "databricks-gemini-2-5-flash": { "id": "databricks-gemini-2-5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "databricks-claude-sonnet-4": { "id": "databricks-claude-sonnet-4", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "databricks-glm-5-2": { "id": "databricks-glm-5-2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.26 } }, "databricks-claude-haiku-4-5": { "id": "databricks-claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "databricks-gpt-5-4-nano": { "id": "databricks-gpt-5-4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "databricks-claude-opus-4-6": { "id": "databricks-claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "databricks-gpt-5-1": { "id": "databricks-gpt-5-1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "databricks-claude-sonnet-4-5": { "id": "databricks-claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "databricks-claude-opus-4-1": { "id": "databricks-claude-opus-4-1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "databricks-gpt-5-5": { "id": "databricks-gpt-5-5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 12.5, "output": 75, "cache_read": 1.25 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "databricks-gemini-3-1-flash-lite": { "id": "databricks-gemini-3-1-flash-lite", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "databricks-gemini-3-pro": { "id": "databricks-gemini-3-pro", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "databricks-kimi-k2-7-code": { "id": "databricks-kimi-k2-7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "databricks-claude-sonnet-4-6": { "id": "databricks-claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "databricks-gpt-oss-20b": { "id": "databricks-gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.05, "output": 0.2 } }, "databricks-gpt-oss-120b": { "id": "databricks-gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.072, "output": 0.28 } }, "databricks-gpt-5-2": { "id": "databricks-gpt-5-2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "databricks-gpt-5-4-mini": { "id": "databricks-gpt-5-4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } } } }, "crof": { "id": "crof", "env": ["CROF_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://crof.ai/v1", "name": "CrofAI", "doc": "https://crof.ai/docs", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.12, "output": 0.21, "cache_read": 0.003 } }, "minimax-m2.5": { "id": "minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.11, "output": 0.95, "cache_read": 0.02, "cache_write": 0.375 } }, "greg-2-ultra": { "id": "greg-2-ultra", "name": "Greg 2 Ultra", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-06-14", "last_updated": "2026-06-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 229376, "output": 229376 }, "cost": { "input": 3, "output": 10, "cache_read": 0.5 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.25, "output": 1.1, "cache_read": 0.05, "cache_write": 0 } }, "deepseek-v4-pro-lightning": { "id": "deepseek-v4-pro-lightning", "name": "DeepSeek V4 Pro Lightning", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.8, "output": 1.6, "cache_read": 0.02 } }, "greg-rp": { "id": "greg-rp", "name": "Greg (Roleplay)", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 229376, "output": 229376 }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "gemma-4-31b-it": { "id": "gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.1, "output": 0.3, "cache_read": 0.02 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.55, "output": 2.25, "cache_read": 0.05 } }, "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.45, "output": 2.15, "cache_read": 0.08, "cache_write": 0 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.35, "output": 0.8, "cache_read": 0.003 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.5, "output": 2.2, "cache_read": 0.08 } }, "greg-2-super": { "id": "greg-2-super", "name": "Greg 2 Super", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-06-14", "last_updated": "2026-06-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 229376, "output": 229376 }, "cost": { "input": 1.5, "output": 5, "cache_read": 0.25 } }, "kimi-k2.5-lightning": { "id": "kimi-k2.5-lightning", "name": "Kimi K2.5 (Lightning)", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "release_date": "2026-02-06", "last_updated": "2026-02-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.35, "output": 1.7, "cache_read": 0.07 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.5, "output": 1.99, "cache_read": 0.05 } }, "qwen3.6-27b": { "id": "qwen3.6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.2, "output": 1.5, "cache_read": 0.04 } }, "qwen3.5-9b": { "id": "qwen3.5-9b", "name": "Qwen3.5 9B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-13", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.04, "output": 0.15, "cache_read": 0.008 } }, "qwen3.5-397b-a17b": { "id": "qwen3.5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.35, "output": 1.75, "cache_read": 0.07 } }, "glm-4.7-flash": { "id": "glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.04, "output": 0.3, "cache_read": 0.008, "cache_write": 0 } }, "mimo-v2.5-pro": { "id": "mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.4, "output": 0.8, "cache_read": 0.003, "tiers": [ { "input": 2, "output": 6, "cache_read": 0.4, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2, "output": 6, "cache_read": 0.4 } } }, "glm-5": { "id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "provider": { "npm": "@ai-sdk/openai-compatible" }, "cost": { "input": 0.48, "output": 1.9, "cache_read": 0.1, "cache_write": 0 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 0.18, "output": 0.35, "cache_read": 0.04 } }, "greg-1-mini": { "id": "greg-1-mini", "name": "Greg 1 Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 229376, "output": 229376 }, "cost": { "input": 0.07, "output": 0.15, "cache_read": 0.01 } } } }, "fastrouter": { "id": "fastrouter", "env": ["FASTROUTER_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://go.fastrouter.ai/api/v1", "name": "FastRouter", "doc": "https://fastrouter.ai/models", "models": { "wanx/wan-v2-6": { "id": "wanx/wan-v2-6", "name": "Wan 2.6", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": true, "limit": { "context": 400000, "output": 0 } }, "moonshotai/kimi-k2": { "id": "moonshotai/kimi-k2", "name": "Kimi K2", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-07-11", "last_updated": "2025-07-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.55, "output": 2.2 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.75, "output": 3.5 } }, "google/imagen-4.0-fast": { "id": "google/imagen-4.0-fast", "name": "Imagen 4 Fast", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-2.5-pro": { "id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.31 } }, "google/gemini-2.5-flash": { "id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.0375 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9 } }, "google/veo3.1-lite": { "id": "google/veo3.1-lite", "name": "Veo 3.1 Lite", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 400000, "output": 0 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.13, "output": 0.38 } }, "google/veo3.1": { "id": "google/veo3.1", "name": "Veo 3.1", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 400000, "output": 0 } }, "google/imagen-4.0-ultra": { "id": "google/imagen-4.0-ultra", "name": "Imagen 4 Ultra", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "imagen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["image"] }, "open_weights": false, "limit": { "context": 480, "output": 0 } }, "google/gemini-3-pro-image-preview": { "id": "google/gemini-3-pro-image-preview", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 32768 }, "cost": { "input": 2, "output": 12 } }, "google/gemini-3.1-flash-image-preview": { "id": "google/gemini-3.1-flash-image-preview", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 65536 }, "cost": { "input": 0.5, "output": 3 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12 } }, "google/veo3.1-fast": { "id": "google/veo3.1-fast", "name": "Veo 3.1 Fast", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 400000, "output": 0 } }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 30000 }, "cost": { "input": 1.25, "output": 2.5 } }, "x-ai/grok-4": { "id": "x-ai/grok-4", "name": "Grok 4", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.75, "cache_write": 15 } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2 } }, "z-ai/glm-5.1": { "id": "z-ai/glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 1.05, "output": 3.5 } }, "z-ai/glm-5": { "id": "z-ai/glm-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.95, "output": 3.15 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10-01", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "openai/gpt-realtime-1.5": { "id": "openai/gpt-realtime-1.5", "name": "GPT Realtime 1.5", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text", "audio", "image"], "output": ["text", "audio"] }, "open_weights": false, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 4, "output": 16 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 0.6 } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-10-01", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2024-10-01", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "openai/gpt-5.5-pro": { "id": "openai/gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.05, "output": 0.2 } }, "openai/gpt-image-2": { "id": "openai/gpt-image-2", "name": "GPT Image 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gpt-image", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 128000, "output": 0 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30 } }, "bytedance/seedance-2": { "id": "bytedance/seedance-2", "name": "Seedance 2", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "seed", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image"], "output": ["video"] }, "open_weights": false, "limit": { "context": 4096, "output": 0 } }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25 } }, "anthropic/claude-opus-4.1": { "id": "anthropic/claude-opus-4.1", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "sarvam/sarvam-105b": { "id": "sarvam/sarvam-105b", "name": "Sarvam 105B", "description": "Flagship Indian-language reasoning model for enterprise multilingual applications", "family": "sarvam", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-09-01", "last_updated": "2025-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.04, "output": 0.16 } }, "sarvam/sarvam-30b": { "id": "sarvam/sarvam-30b", "name": "Sarvam 30B", "description": "Efficient Indian-language reasoning model for chat, coding, and multilingual work", "family": "sarvam", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-18", "last_updated": "2026-02-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.02, "output": 0.1 } }, "deepseek-ai/deepseek-r1-distill-llama-70b": { "id": "deepseek-ai/deepseek-r1-distill-llama-70b", "name": "DeepSeek R1 Distill Llama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-10", "release_date": "2025-01-23", "last_updated": "2025-01-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.03, "output": 0.14 } }, "qwen/qwen3-coder": { "id": "qwen/qwen3-coder", "name": "Qwen3 Coder", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 66536 }, "cost": { "input": 0.3, "output": 1.2 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.74, "output": 3.48 } }, "minimax/minimax-m2.7-highspeed": { "id": "minimax/minimax-m2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "leonardo-ai/lucid-origin": { "id": "leonardo-ai/lucid-origin", "name": "Lucid Origin", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "lucid", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 4096, "output": 0 } }, "leonardo-ai/lucid-realism": { "id": "leonardo-ai/lucid-realism", "name": "Lucid Realism", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "lucid", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text", "image"], "output": ["image"] }, "open_weights": false, "limit": { "context": 4096, "output": 0 } } } }, "abliteration-ai": { "id": "abliteration-ai", "env": ["ABLIT_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.abliteration.ai/v1", "name": "abliteration.ai", "doc": "https://docs.abliteration.ai/models", "models": { "abliterated-model": { "id": "abliterated-model", "name": "Abliterated Model", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-01-06", "last_updated": "2026-01-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 150000, "input": 150000, "output": 8192 }, "cost": { "input": 3, "output": 3 } } } }, "xpersona": { "id": "xpersona", "env": ["XPERSONA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://www.xpersona.co/v1", "name": "Xpersona", "doc": "https://www.xpersona.co/docs", "models": { "xpersona-gpt-5.5": { "id": "xpersona-gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-12-30", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 18, "reasoning": 18, "cache_read": 0.3 } }, "xpersona-frieren-coder": { "id": "xpersona-frieren-coder", "name": "Xpersona Frieren 1", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-30", "release_date": "2026-05-01", "last_updated": "2026-05-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.5, "output": 6, "reasoning": 6, "cache_read": 0.15 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 3, "output": 18, "reasoning": 18, "cache_read": 0.3 } } } }, "azure-cognitive-services": { "id": "azure-cognitive-services", "env": ["AZURE_COGNITIVE_SERVICES_RESOURCE_NAME", "AZURE_COGNITIVE_SERVICES_API_KEY"], "npm": "@ai-sdk/azure", "name": "Azure Cognitive Services", "doc": "https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models", "models": { "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-24", "last_updated": "2025-08-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 Nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-12-31", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "claude-opus-4-1": { "id": "claude-opus-4-1", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models", "shape": "completions" }, "cost": { "input": 0.6, "output": 3 } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-31", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 Mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models", "shape": "completions" }, "cost": { "input": 0.95, "output": 4 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "provider": { "npm": "@ai-sdk/anthropic", "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "gpt-5.4-pro": { "id": "gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "tiers": [ { "input": 60, "output": 270, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 60, "output": 270 } } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "image", "audio"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "meta-llama-3-70b-instruct": { "id": "meta-llama-3-70b-instruct", "name": "Meta-Llama-3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 2.68, "output": 3.54 } }, "phi-3-mini-4k-instruct": { "id": "phi-3-mini-4k-instruct", "name": "Phi-3-mini-instruct (4k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 1024 }, "cost": { "input": 0.13, "output": 0.52 } }, "phi-3-mini-128k-instruct": { "id": "phi-3-mini-128k-instruct", "name": "Phi-3-mini-instruct (128k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.13, "output": 0.52 } }, "phi-4-mini-reasoning": { "id": "phi-4-mini-reasoning", "name": "Phi-4-mini-reasoning", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.075, "output": 0.3 } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "llama-4-maverick-17b-128e-instruct-fp8": { "id": "llama-4-maverick-17b-128e-instruct-fp8", "name": "Llama 4 Maverick 17B 128E Instruct FP8", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.25, "output": 1 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.58, "output": 1.68 } }, "gpt-5-codex": { "id": "gpt-5-codex", "name": "GPT-5-Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "model-router": { "id": "model-router", "name": "Model Router", "description": "Automatic model router for matching prompts to suitable backends and budgets", "family": "model-router", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2025-05-19", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.14, "output": 0 } }, "gpt-4o-mini": { "id": "gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.075 } }, "phi-4-multimodal": { "id": "phi-4-multimodal", "name": "Phi-4-multimodal", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "phi", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.08, "output": 0.32, "input_audio": 4 } }, "phi-4-reasoning-plus": { "id": "phi-4-reasoning-plus", "name": "Phi-4-reasoning-plus", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.125, "output": 0.5 } }, "codestral-2501": { "id": "codestral-2501", "name": "Codestral 25.01", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "codestral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 256000 }, "cost": { "input": 0.3, "output": 0.9 } }, "cohere-embed-v3-english": { "id": "cohere-embed-v3-english", "name": "Embed v3 English", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "cohere-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-11-07", "last_updated": "2023-11-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 1024 }, "cost": { "input": 0.1, "output": 0 } }, "phi-3-medium-4k-instruct": { "id": "phi-3-medium-4k-instruct", "name": "Phi-3-medium-instruct (4k)", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 4096, "output": 1024 }, "cost": { "input": 0.17, "output": 0.68 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.01 } }, "gpt-4-turbo": { "id": "gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "gpt-4.1-mini": { "id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.1 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.03 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "ministral-3b": { "id": "ministral-3b", "name": "Ministral 3B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-03", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.04, "output": 0.04 } }, "llama-3.2-90b-vision-instruct": { "id": "llama-3.2-90b-vision-instruct", "name": "Llama-3.2-90B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 2.04, "output": 2.04 } }, "deepseek-v3.1": { "id": "deepseek-v3.1", "name": "DeepSeek-V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.56, "output": 1.68 } }, "o1": { "id": "o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 15, "output": 60, "cache_read": 7.5 } }, "gpt-4.1-nano": { "id": "gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "grok-4-fast-reasoning": { "id": "grok-4-fast-reasoning", "name": "Grok 4 Fast (Reasoning)", "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-07", "release_date": "2025-09-19", "last_updated": "2025-09-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "output": 30000 }, "cost": { "input": 0.2, "output": 0.5, "cache_read": 0.05 } }, "phi-4": { "id": "phi-4", "name": "Phi-4", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.125, "output": 0.5 } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "GPT-5.1 Codex Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "meta-llama-3.1-8b-instruct": { "id": "meta-llama-3.1-8b-instruct", "name": "Meta-Llama-3.1-8B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.61 } }, "gpt-3.5-turbo-0301": { "id": "gpt-3.5-turbo-0301", "name": "GPT-3.5 Turbo 0301", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-03-01", "last_updated": "2023-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 1.5, "output": 2 } }, "deepseek-v3.2-speciale": { "id": "deepseek-v3.2-speciale", "name": "DeepSeek-V3.2-Speciale", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.58, "output": 1.68 } }, "text-embedding-3-small": { "id": "text-embedding-3-small", "name": "text-embedding-3-small", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 1536 }, "cost": { "input": 0.02, "output": 0 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "mistral-large-2411": { "id": "mistral-large-2411", "name": "Mistral Large 24.11", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 2, "output": 6 } }, "gpt-4-turbo-vision": { "id": "gpt-4-turbo-vision", "name": "GPT-4 Turbo Vision", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.125 } }, "o3-mini": { "id": "o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "text-embedding-ada-002": { "id": "text-embedding-ada-002", "name": "text-embedding-ada-002", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2022-12-15", "last_updated": "2022-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 1536 }, "cost": { "input": 0.1, "output": 0 } }, "meta-llama-3.1-70b-instruct": { "id": "meta-llama-3.1-70b-instruct", "name": "Meta-Llama-3.1-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 2.68, "output": 3.54 } }, "mistral-medium-2505": { "id": "mistral-medium-2505", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.4, "output": 2 } }, "phi-4-reasoning": { "id": "phi-4-reasoning", "name": "Phi-4-reasoning", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "phi", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32000, "output": 4096 }, "cost": { "input": 0.125, "output": 0.5 } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt-codex", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "image", "audio"] }, "open_weights": false, "limit": { "context": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-3.5-turbo-0613": { "id": "gpt-3.5-turbo-0613", "name": "GPT-3.5 Turbo 0613", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-06-13", "last_updated": "2023-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 3, "output": 4 } }, "cohere-embed-v3-multilingual": { "id": "cohere-embed-v3-multilingual", "name": "Embed v3 Multilingual", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "cohere-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-11-07", "last_updated": "2023-11-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 512, "output": 1024 }, "cost": { "input": 0.1, "output": 0 } }, "gpt-3.5-turbo-0125": { "id": "gpt-3.5-turbo-0125", "name": "GPT-3.5 Turbo 0125", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 0.5, "output": 1.5 } }, "phi-3-small-8k-instruct": { "id": "phi-3-small-8k-instruct", "name": "Phi-3-small-instruct (8k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0.15, "output": 0.6 } }, "deepseek-r1": { "id": "deepseek-r1", "name": "DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 1.35, "output": 5.4 } }, "kimi-k2-thinking": { "id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-12-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.6, "output": 2.5, "cache_read": 0.15 } }, "gpt-5.1-chat": { "id": "gpt-5.1-chat", "name": "GPT-5.1 Chat", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-14", "last_updated": "2025-11-14", "modalities": { "input": ["text", "image", "audio"], "output": ["text", "image", "audio"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "llama-3.3-70b-instruct": { "id": "llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.71, "output": 0.71 } }, "llama-4-scout-17b-16e-instruct": { "id": "llama-4-scout-17b-16e-instruct", "name": "Llama 4 Scout 17B 16E Instruct", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.2, "output": 0.78 } }, "gpt-3.5-turbo-1106": { "id": "gpt-3.5-turbo-1106", "name": "GPT-3.5 Turbo 1106", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-11-06", "last_updated": "2023-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "output": 16384 }, "cost": { "input": 1, "output": 2 } }, "phi-4-mini": { "id": "phi-4-mini", "name": "Phi-4-mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-10", "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.075, "output": 0.3 } }, "cohere-command-r-plus-08-2024": { "id": "cohere-command-r-plus-08-2024", "name": "Command R+", "description": "Cohere's RAG workhorse for long-context enterprise search and tool use", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 2.5, "output": 10 } }, "meta-llama-3.1-405b-instruct": { "id": "meta-llama-3.1-405b-instruct", "name": "Meta-Llama-3.1-405B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 5.33, "output": 16 } }, "gpt-4-32k": { "id": "gpt-4-32k", "name": "GPT-4 32K", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-03-14", "last_updated": "2023-03-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 60, "output": 120 } }, "phi-3-medium-128k-instruct": { "id": "phi-3-medium-128k-instruct", "name": "Phi-3-medium-instruct (128k)", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.17, "output": 0.68 } }, "o4-mini": { "id": "o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.275 } }, "gpt-4": { "id": "gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-03-14", "last_updated": "2023-03-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 60, "output": 120 } }, "gpt-4o": { "id": "gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10, "cache_read": 1.25 } }, "cohere-command-r-08-2024": { "id": "cohere-command-r-08-2024", "name": "Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4000 }, "cost": { "input": 0.15, "output": 0.6 } }, "gpt-5-pro": { "id": "gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "output": 272000 }, "cost": { "input": 15, "output": 120 } }, "llama-3.2-11b-vision-instruct": { "id": "llama-3.2-11b-vision-instruct", "name": "Llama-3.2-11B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "cost": { "input": 0.37, "output": 0.37 } }, "cohere-command-a": { "id": "cohere-command-a", "name": "Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 8000 }, "cost": { "input": 2.5, "output": 10 } }, "gpt-5-chat": { "id": "gpt-5-chat", "name": "GPT-5 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "knowledge": "2024-10-24", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "cohere-embed-v-4-0": { "id": "cohere-embed-v-4-0", "name": "Embed v4", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "cohere-embed", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 1536 }, "cost": { "input": 0.12, "output": 0 } }, "mistral-nemo": { "id": "mistral-nemo", "name": "Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.15 } }, "phi-3.5-mini-instruct": { "id": "phi-3.5-mini-instruct", "name": "Phi-3.5-mini-instruct", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-08-20", "last_updated": "2024-08-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.13, "output": 0.52 } }, "o1-mini": { "id": "o1-mini", "name": "o1-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-09-12", "last_updated": "2024-09-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 65536 }, "cost": { "input": 1.1, "output": 4.4, "cache_read": 0.55 } }, "text-embedding-3-large": { "id": "text-embedding-3-large", "name": "text-embedding-3-large", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8191, "output": 3072 }, "cost": { "input": 0.13, "output": 0 } }, "mistral-small-2503": { "id": "mistral-small-2503", "name": "Mistral Small 3.1", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2025-03-01", "last_updated": "2025-03-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.3 } }, "meta-llama-3-8b-instruct": { "id": "meta-llama-3-8b-instruct", "name": "Meta-Llama-3-8B-Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-12", "release_date": "2024-04-18", "last_updated": "2024-04-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 2048 }, "cost": { "input": 0.3, "output": 0.61 } }, "phi-3-small-128k-instruct": { "id": "phi-3-small-128k-instruct", "name": "Phi-3-small-instruct (128k)", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-04-23", "last_updated": "2024-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.15, "output": 0.6 } }, "deepseek-v3-0324": { "id": "deepseek-v3-0324", "name": "DeepSeek-V3-0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 1.14, "output": 4.56 } }, "o3": { "id": "o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "gpt-5.2-chat": { "id": "gpt-5.2-chat", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 16384 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "deepseek-r1-0528": { "id": "deepseek-r1-0528", "name": "DeepSeek-R1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 163840 }, "cost": { "input": 1.35, "output": 5.4 } }, "gpt-3.5-turbo-instruct": { "id": "gpt-3.5-turbo-instruct", "name": "GPT-3.5 Turbo Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2021-08", "release_date": "2023-09-21", "last_updated": "2023-09-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "output": 4096 }, "cost": { "input": 1.5, "output": 2 } }, "phi-3.5-moe-instruct": { "id": "phi-3.5-moe-instruct", "name": "Phi-3.5-MoE-instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2023-10", "release_date": "2024-08-20", "last_updated": "2024-08-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.16, "output": 0.64 } }, "codex-mini": { "id": "codex-mini", "name": "Codex Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": false, "knowledge": "2024-04", "release_date": "2025-05-16", "last_updated": "2025-05-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 100000 }, "cost": { "input": 1.5, "output": 6, "cache_read": 0.375 } } } }, "baseten": { "id": "baseten", "env": ["BASETEN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://inference.baseten.co/v1", "name": "Baseten", "doc": "https://docs.baseten.co/inference/model-apis/overview", "models": { "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "moonshotai/Kimi-K2.5": { "id": "moonshotai/Kimi-K2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-01-30", "last_updated": "2026-02-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.12 } }, "moonshotai/Kimi-K2.7-Code": { "id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262000, "output": 262000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "OpenAI GPT 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128072, "output": 128072 }, "cost": { "input": 0.1, "output": 0.5 } }, "nvidia/Nemotron-120B-A12B": { "id": "nvidia/Nemotron-120B-A12B", "name": "Nemotron Super", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2026-02", "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 202800 }, "cost": { "input": 0.3, "output": 0.75, "cache_read": 0.06 } }, "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", "name": "Nemotron Ultra", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 202800 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.12 } }, "zai-org/GLM-5": { "id": "zai-org/GLM-5", "name": "GLM 5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 202800 }, "cost": { "input": 0.95, "output": 3.15, "cache_read": 0.2 } }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM 4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 200000 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.12 } }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202720, "output": 202720 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.3 } }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202800, "output": 202800 }, "cost": { "input": 1.3, "output": 4.3, "cache_read": 0.26 } }, "deepseek-ai/DeepSeek-V3.1": { "id": "deepseek-ai/DeepSeek-V3.1", "name": "DeepSeek V3.1", "description": "Legacy model retained for compatibility with older integrations", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-08-25", "last_updated": "2025-08-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 164000, "output": 131000 }, "status": "deprecated", "cost": { "input": 0.5, "output": 1.5 } }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "Deepseek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.74, "output": 3.48, "cache_read": 0.145 } }, "MiniMaxAI/MiniMax-M2.5": { "id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "Legacy model retained for compatibility with older integrations", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2026-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204000, "output": 204000 }, "status": "deprecated", "cost": { "input": 0.3, "output": 1.2 } } } }, "atomic-chat": { "id": "atomic-chat", "env": ["ATOMIC_CHAT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "http://127.0.0.1:1337/v1", "name": "Atomic Chat", "doc": "https://atomic.chat", "models": { "gemma-4-E4B-it-IQ4_XS": { "id": "gemma-4-E4B-it-IQ4_XS", "name": "Gemma 4 E4B Instruct (IQ4_XS)", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "Meta-Llama-3_1-8B-Instruct-GGUF": { "id": "Meta-Llama-3_1-8B-Instruct-GGUF", "name": "Meta Llama 3.1 8B Instruct (GGUF)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 4096 }, "cost": { "input": 0, "output": 0 } }, "Qwen3_5-9B-MLX-4bit": { "id": "Qwen3_5-9B-MLX-4bit", "name": "Qwen 3.5 9B (MLX 4-bit)", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-05", "last_updated": "2026-04-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "gemma-4-E4B-it-MLX-4bit": { "id": "gemma-4-E4B-it-MLX-4bit", "name": "Gemma 4 E4B Instruct (MLX 4-bit)", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "Qwen3_5-9B-Q4_K_M": { "id": "Qwen3_5-9B-Q4_K_M", "name": "Qwen 3.5 9B (Q4_K_M)", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-05", "last_updated": "2026-04-04", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 32768, "output": 8192 }, "cost": { "input": 0, "output": 0 } } } }, "meta": { "id": "meta", "env": ["META_MODEL_API_KEY"], "npm": "@ai-sdk/openai", "api": "https://api.meta.ai/v1", "name": "Meta", "doc": "https://dev.meta.ai/docs", "models": { "muse-spark-1.1": { "id": "muse-spark-1.1", "name": "Muse Spark 1.1", "description": "Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.", "family": "muse", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 1.25, "output": 4.25, "cache_read": 0.15 } } } }, "routing-run": { "id": "routing-run", "env": ["ROUTING_RUN_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.routing.run/v1", "name": "routing.run", "doc": "https://docs.routing.run/api-reference/models", "models": { "kimi-k2.6-nitro": { "id": "kimi-k2.6-nitro", "name": "Kimi K2.6 Nitro", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.275, "output": 1.1 } }, "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.112, "output": 0.224 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.275, "output": 1.1 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 0.348, "output": 0.696 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.8, "output": 2.4 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 32000 }, "cost": { "input": 5, "output": 25 } }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 }, "cost": { "input": 0.7, "output": 4.2 } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 }, "cost": { "input": 1.5, "output": 9 } }, "glm-5.2-nitro": { "id": "glm-5.2-nitro", "name": "GLM 5.2 Nitro", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.8, "output": 2.4 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.275, "output": 1.1 } }, "qwen3.5-9b": { "id": "qwen3.5-9b", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32000 }, "cost": { "input": 0.16, "output": 0.48 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "nemotron-3-ultra": { "id": "nemotron-3-ultra", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32000 }, "cost": { "input": 0.1, "output": 0.1 } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15 } }, "kimi-k2.7-code-nitro": { "id": "kimi-k2.7-code-nitro", "name": "Kimi K2.7 Code Nitro", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 0.275, "output": 1.1 } } } }, "aihubmix": { "id": "aihubmix", "env": ["AIHUBMIX_API_KEY"], "npm": "@aihubmix/ai-sdk-provider", "name": "AIHubMix", "doc": "https://docs.aihubmix.com", "models": { "coding-minimax-m2.7": { "id": "coding-minimax-m2.7", "name": "Coding MiniMax M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128100 }, "cost": { "input": 0.2, "output": 0.2 } }, "alicloud-glm-5.1": { "id": "alicloud-glm-5.1", "name": "GLM-5.1 (Alibaba Cloud)", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.84, "output": 3.38, "cache_read": 0.169, "cache_write": 1.05625 } }, "claude-sonnet-4-6-think": { "id": "claude-sonnet-4-6-think", "name": "Claude Sonnet 4.6 Thinking", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "gemini-3.1-flash-lite": { "id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "cache_write": 1 } }, "alicloud-deepseek-v4-flash": { "id": "alicloud-deepseek-v4-flash", "name": "DeepSeek V4 Flash (Alibaba Cloud)", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "xiaomi-mimo-v2.5-pro": { "id": "xiaomi-mimo-v2.5-pro", "name": "Xiaomi MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo-v2.5-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-05-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1.1, "output": 3.3, "cache_read": 0.22, "tiers": [ { "input": 2.2, "output": 6.6, "cache_read": 0.44, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 2.2, "output": 6.6, "cache_read": 0.44 } } }, "doubao-seed-2-0-code-preview": { "id": "doubao-seed-2-0-code-preview", "name": "Doubao Seed 2.0 Code Preview", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.48, "output": 2.41, "cache_read": 0.09644, "tiers": [ { "input": 0.72, "output": 3.62, "cache_read": 0.144656, "tier": { "type": "context", "size": 32000 } }, { "input": 1.45, "output": 7.23, "cache_read": 0.28932, "tier": { "type": "context", "size": 128000 } } ] } }, "coding-xiaomi-mimo-v2.5-pro": { "id": "coding-xiaomi-mimo-v2.5-pro", "name": "Coding Xiaomi MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo-v2.5-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-05-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.2, "output": 0.6, "cache_read": 0.04, "tiers": [ { "input": 0.4, "output": 1.2, "cache_read": 0.08, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.4, "output": 1.2, "cache_read": 0.08 } } }, "doubao-seed-2-0-pro": { "id": "doubao-seed-2-0-pro", "name": "Doubao Seed 2.0 Pro", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.48, "output": 2.41, "cache_read": 0.09644, "tiers": [ { "input": 0.72, "output": 3.62, "cache_read": 0.144656, "tier": { "type": "context", "size": 32000 } }, { "input": 1.45, "output": 7.23, "cache_read": 0.28932, "tier": { "type": "context", "size": 128000 } } ] } }, "deep-deepseek-v4-flash": { "id": "deep-deepseek-v4-flash", "name": "DeepSeek V4 Flash (DeepSeek)", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.154, "output": 0.308, "cache_read": 0.0308 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991000, "output": 64000 }, "cost": { "input": 0.282, "output": 1.128, "cache_read": 0.0564, "cache_write": 0.3525 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-03-20", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "grok-4.3": { "id": "grok-4.3", "name": "Grok 4.3", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 1000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [ { "input": 2.5, "output": 5, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 5, "cache_read": 0.4 } } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "max": 262144 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991000, "output": 64000 }, "cost": { "input": 1.69, "output": 5.07, "cache_read": 0.169, "cache_write": 2.1125 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-03-20", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "xiaomi-mimo-v2.5-free": { "id": "xiaomi-mimo-v2.5-free", "name": "Xiaomi MiMo-V2.5 (free)", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo-v2.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-05-13", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "claude-opus-4-7-think": { "id": "claude-opus-4-7-think", "name": "Claude Opus 4.7 Thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "coding-glm-5.1-free": { "id": "coding-glm-5.1-free", "name": "Coding GLM 5.1 (free)", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm-free", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-11", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "xiaomi-mimo-v2.5-pro-free": { "id": "xiaomi-mimo-v2.5-pro-free", "name": "Xiaomi MiMo-V2.5-Pro (free)", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo-v2.5-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-05-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "xiaomi-mimo-v2.5": { "id": "xiaomi-mimo-v2.5", "name": "Xiaomi MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo-v2.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-05-13", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.44, "output": 2.2, "cache_read": 0.088, "tiers": [ { "input": 0.88, "output": 4.4, "cache_read": 0.176, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.88, "output": 4.4, "cache_read": 0.176 } } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 1.1268, "output": 3.9438, "cache_read": 0.2817 } }, "qwen3.6-flash": { "id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991000, "output": 64000 }, "cost": { "input": 0.17, "output": 1.01, "cache_read": 0.0169, "cache_write": 0.21125, "tiers": [ { "input": 0.68, "output": 4.06, "cache_read": 0.0676, "cache_write": 0.845, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.68, "output": 4.06, "cache_read": 0.0676, "cache_write": 0.845 } } }, "coding-xiaomi-mimo-v2.5": { "id": "coding-xiaomi-mimo-v2.5", "name": "Coding Xiaomi MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo-v2.5", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-05-13", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0.08, "output": 0.4, "cache_read": 0.016, "tiers": [ { "input": 0.16, "output": 0.8, "cache_read": 0.032, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 0.16, "output": 0.8, "cache_read": 0.032 } } }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gemini-3.1-pro-preview-customtools": { "id": "gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "doubao-seed-2-0-mini-260428": { "id": "doubao-seed-2-0-mini-260428", "name": "Doubao Seed 2.0 Mini 260428", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.03, "output": 0.28, "cache_read": 0.00564, "input_audio": 0.423, "tiers": [ { "input": 0.06, "output": 0.56, "cache_read": 0.01128, "input_audio": 0.846, "tier": { "type": "context", "size": 32000 } }, { "input": 0.11, "output": 1.13, "cache_read": 0.02256, "input_audio": 1.692, "tier": { "type": "context", "size": 128000 } } ] } }, "doubao-seed-2-0-lite-260428": { "id": "doubao-seed-2-0-lite-260428", "name": "Doubao Seed 2.0 Lite 260428", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 128000 }, "cost": { "input": 0.08, "output": 0.51, "cache_read": 0.01692, "input_audio": 1.269, "tiers": [ { "input": 0.13, "output": 0.76, "cache_read": 0.02536, "input_audio": 1.902, "tier": { "type": "context", "size": 32000 } }, { "input": 0.25, "output": 1.52, "cache_read": 0.05072, "input_audio": 3.804, "tier": { "type": "context", "size": 128000 } } ] } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "minimax-m2.7": { "id": "minimax-m2.7", "name": "MiniMax M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } }, "claude-opus-4-6-think": { "id": "claude-opus-4-6-think", "name": "Claude Opus 4.6 Thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "kimi-k2.5": { "id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.6, "output": 3, "cache_read": 0.1 } }, "coding-glm-5.1": { "id": "coding-glm-5.1", "name": "Coding GLM 5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-11", "last_updated": "2026-04-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.06, "output": 0.22, "cache_read": 0.013 } }, "coding-minimax-m2.7-free": { "id": "coding-minimax-m2.7-free", "name": "Coding MiniMax M2.7 (Free)", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax-free", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128100 }, "cost": { "input": 0, "output": 0 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 5, "output": 30, "cache_read": 0.5 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "kimi-k2.6": { "id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.16 } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "deep-deepseek-v4-pro": { "id": "deep-deepseek-v4-pro", "name": "DeepSeek V4 Pro (DeepSeek)", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.478, "output": 0.956, "cache_read": 0.004302 } }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM 5 Vision Turbo", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glmv", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-09", "last_updated": "2026-05-09", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.7042, "output": 3.09848, "cache_read": 0.169008 } }, "zai-glm-5.1": { "id": "zai-glm-5.1", "name": "GLM-5.1 (Z.ai)", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0.845, "output": 3.38, "cache_read": 0.183112 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "interleaved": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "tiers": [ { "input": 0.5, "output": 3, "cache_read": 0.05, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 0.5, "output": 3, "cache_read": 0.05 } } }, "coding-minimax-m2.7-highspeed": { "id": "coding-minimax-m2.7-highspeed", "name": "Coding MiniMax M2.7 Highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 128100 }, "cost": { "input": 0.2, "output": 0.2 } }, "claude-opus-4-8-think": { "id": "claude-opus-4-8-think", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "qwen3.6-max-preview": { "id": "qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "qwen3.6", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-05-09", "last_updated": "2026-05-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 240000, "output": 64000 }, "cost": { "input": 1.27, "output": 7.61, "cache_read": 0.1268, "cache_write": 1.585, "tiers": [ { "input": 2.11, "output": 12.67, "cache_read": 0.2112, "cache_write": 2.64, "tier": { "type": "context", "size": 128000 } } ] } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "qwen3.6-plus": { "id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-05-09", "last_updated": "2026-05-09", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991000, "output": 64000 }, "cost": { "input": 0.28, "output": 1.69, "cache_read": 0.0282, "cache_write": 0.3525, "tiers": [ { "input": 1.13, "output": 6.77, "cache_read": 0.1128, "cache_write": 1.41, "tier": { "type": "context", "size": 256000 } } ], "context_over_200k": { "input": 1.13, "output": 6.77, "cache_read": 0.1128, "cache_write": 1.41 } } }, "alicloud-deepseek-v4-pro": { "id": "alicloud-deepseek-v4-pro", "name": "DeepSeek V4 Pro (Alibaba Cloud)", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 1.69, "output": 3.38, "cache_read": 0.13 } }, "gpt-5.1": { "id": "gpt-5.1", "name": "GPT-5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.13 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "experimental": { "modes": { "fast": { "cost": { "input": 12.5, "output": 75, "cache_read": 1.25 }, "provider": { "body": { "service_tier": "priority" } } } } }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "google-vertex": { "id": "google-vertex", "env": ["GOOGLE_VERTEX_PROJECT", "GOOGLE_VERTEX_LOCATION", "GOOGLE_APPLICATION_CREDENTIALS"], "npm": "@ai-sdk/google-vertex", "name": "Vertex", "doc": "https://cloud.google.com/vertex-ai/generative-ai/docs/models", "models": { "gemini-2.5-pro-tts": { "id": "gemini-2.5-pro-tts", "name": "Gemini 2.5 Pro TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gemini-pro", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-01", "release_date": "2025-09-30", "last_updated": "2025-12-10", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 1, "output": 20 } }, "claude-haiku-4-5@20251001": { "id": "claude-haiku-4-5@20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "gemini-3.1-flash-lite": { "id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "gemini-3.1-flash-image": { "id": "gemini-3.1-flash-image", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.5, "output": 60 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "gemini-2.5-flash-tts": { "id": "gemini-2.5-flash-tts", "name": "Gemini 2.5 Flash TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gemini-flash", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-01", "release_date": "2025-09-30", "last_updated": "2025-12-10", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": false, "limit": { "context": 32768, "output": 16384 }, "cost": { "input": 0.5, "output": 10 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075, "cache_write": 0.383 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "claude-opus-4@20250514": { "id": "claude-opus-4@20250514", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "claude-opus-4-1@20250805": { "id": "claude-opus-4-1@20250805", "name": "Claude Opus 4.1", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "gemini-embedding-001": { "id": "gemini-embedding-001", "name": "Gemini Embedding 001", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "gemini", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-05", "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2048, "output": 1 }, "cost": { "input": 0.15, "output": 0 } }, "claude-opus-4-5@20251101": { "id": "claude-opus-4-5@20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-3-5-haiku@20241022": { "id": "claude-3-5-haiku@20241022", "name": "Claude Haiku 3.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 8192 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 0.8, "output": 4, "cache_read": 0.08, "cache_write": 1 } }, "gemini-3.1-pro-preview-customtools": { "id": "gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "gemini-flash-lite-latest": { "id": "gemini-flash-lite-latest", "name": "Gemini Flash-Lite Latest", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.025 } }, "gemini-2.5-flash-image": { "id": "gemini-2.5-flash-image", "name": "Nano Banana", "description": "Nano Banana image model for fast generation, edits, and character-consistent assets", "family": "gemini-flash", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 32768, "output": 32768 }, "cost": { "input": 0.3, "output": 30 } }, "claude-sonnet-4@20250514": { "id": "claude-sonnet-4@20250514", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "status": "deprecated", "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 512, "max": 24576 } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4, "cache_read": 0.01, "input_audio": 0.3 } }, "claude-opus-4-7@default": { "id": "claude-opus-4-7@default", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "claude-sonnet-4-5@20250929": { "id": "claude-sonnet-4-5@20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-sonnet-5@default": { "id": "claude-sonnet-5@default", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1 } }, "claude-opus-4-6@default": { "id": "claude-opus-4-6@default", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "gemini-flash-latest": { "id": "gemini-flash-latest", "name": "Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.075, "cache_write": 0.383 } }, "claude-opus-4-8@default": { "id": "claude-opus-4-8@default", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [ { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5 } } }, "gemini-3-pro-image": { "id": "gemini-3-pro-image", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 65536, "output": 32768 }, "cost": { "input": 2, "output": 120 } }, "claude-sonnet-4-6@default": { "id": "claude-sonnet-4-6@default", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "provider": { "npm": "@ai-sdk/google-vertex/anthropic" }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [ { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5 } } }, "gemini-3.1-flash-lite-preview": { "id": "gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "status": "deprecated", "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "moonshotai/kimi-k2-thinking-maas": { "id": "moonshotai/kimi-k2-thinking-maas", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 0.6, "output": 2.5 } }, "openai/gpt-oss-120b-maas": { "id": "openai/gpt-oss-120b-maas", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.09, "output": 0.36 } }, "openai/gpt-oss-20b-maas": { "id": "openai/gpt-oss-20b-maas", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.07, "output": 0.25 } }, "zai-org/glm-4.7-maas": { "id": "zai-org/glm-4.7-maas", "name": "GLM-4.7", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-06", "last_updated": "2026-01-06", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 0.6, "output": 2.2 } }, "zai-org/glm-5-maas": { "id": "zai-org/glm-5-maas", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.1 } }, "deepseek-ai/deepseek-v3.1-maas": { "id": "deepseek-ai/deepseek-v3.1-maas", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 32768 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 0.6, "output": 1.7 } }, "deepseek-ai/deepseek-v3.2-maas": { "id": "deepseek-ai/deepseek-v3.2-maas", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-17", "last_updated": "2026-04-04", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": true, "limit": { "context": 163840, "output": 65536 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 0.56, "output": 1.68, "cache_read": 0.056 } }, "qwen/qwen3-235b-a22b-instruct-2507-maas": { "id": "qwen/qwen3-235b-a22b-instruct-2507-maas", "name": "Qwen3 235B A22B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-13", "last_updated": "2025-08-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 16384 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 0.22, "output": 0.88 } }, "meta/llama-3.3-70b-instruct-maas": { "id": "meta/llama-3.3-70b-instruct-maas", "name": "Llama 3.3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 8192 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 0.72, "output": 0.72 } }, "meta/llama-4-maverick-17b-128e-instruct-maas": { "id": "meta/llama-4-maverick-17b-128e-instruct-maas", "name": "Llama 4 Maverick 17B 128E Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 8192 }, "provider": { "npm": "@ai-sdk/openai-compatible", "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi" }, "cost": { "input": 0.35, "output": 1.15 } } } }, "nano-gpt": { "id": "nano-gpt", "env": ["NANO_GPT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://nano-gpt.com/api/v1", "name": "NanoGPT", "doc": "https://docs.nano-gpt.com", "models": { "step-3": { "id": "step-3", "name": "Step-3", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-31", "last_updated": "2025-07-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 8192 }, "cost": { "input": 0.2499, "output": 0.6494 } }, "qwen3.5-35b-a3b:thinking": { "id": "qwen3.5-35b-a3b:thinking", "name": "Qwen3.5 35B A3B Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.225, "output": 1.8 } }, "glm-4.1v-thinking-flashx": { "id": "glm-4.1v-thinking-flashx", "name": "GLM 4.1V Thinking FlashX", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "input": 64000, "output": 8192 }, "cost": { "input": 0.3, "output": 0.3 } }, "ernie-x1.1-preview": { "id": "ernie-x1.1-preview", "name": "ERNIE X1.1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-10", "last_updated": "2025-09-10", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "input": 64000, "output": 8192 }, "cost": { "input": 0.15, "output": 0.6 } }, "qwen25-vl-72b-instruct": { "id": "qwen25-vl-72b-instruct", "name": "Qwen25 VL 72b", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-10", "last_updated": "2025-05-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 32768 }, "cost": { "input": 0.69989, "output": 0.69989 } }, "gemini-2.0-pro-exp-02-05": { "id": "gemini-2.0-pro-exp-02-05", "name": "Gemini 2.0 Pro 0205", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-05", "last_updated": "2025-02-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2097152, "input": 2097152, "output": 8192 }, "cost": { "input": 1.989, "output": 7.956 } }, "doubao-seed-2-0-lite-260215": { "id": "doubao-seed-2-0-lite-260215", "name": "Doubao Seed 2.0 Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32000 }, "cost": { "input": 0.1462, "output": 0.8738 } }, "Qwen3.5-27B-Writer-V2-Derestricted": { "id": "Qwen3.5-27B-Writer-V2-Derestricted", "name": "Qwen3.5 27B Writer V2 Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "claude-opus-4-thinking:8192": { "id": "claude-opus-4-thinking:8192", "name": "Claude 4 Opus Thinking (8K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "qwen3-vl-235b-a22b-thinking": { "id": "qwen3-vl-235b-a22b-thinking", "name": "Qwen3 VL 235B A22B Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.5, "output": 6 } }, "glm-4-air-0111": { "id": "glm-4-air-0111", "name": "GLM 4 Air 0111", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-11", "last_updated": "2025-01-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 0.1394, "output": 0.1394 } }, "Qwen3.5-27B-Queen-Derestricted": { "id": "Qwen3.5-27B-Queen-Derestricted", "name": "Qwen3.5 27B Queen Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "gemini-2.5-pro-preview-03-25": { "id": "gemini-2.5-pro-preview-03-25", "name": "Gemini 2.5 Pro Preview 0325", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2.5, "output": 10 } }, "brave-research": { "id": "brave-research", "name": "Brave (Research)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-03-02", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 5, "output": 5 } }, "qwen-plus": { "id": "qwen-plus", "name": "Qwen Plus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 995904, "input": 995904, "output": 32768 }, "cost": { "input": 0.3995, "output": 1.2002 } }, "doubao-seed-2-0-mini-260215": { "id": "doubao-seed-2-0-mini-260215", "name": "Doubao Seed 2.0 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32000 }, "cost": { "input": 0.0493, "output": 0.4845 } }, "Qwen3.5-27B-BlueStar-v3-Derestricted-Lite": { "id": "Qwen3.5-27B-BlueStar-v3-Derestricted-Lite", "name": "Qwen3.5 27B BlueStar v3 Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "deepseek-v3-0324": { "id": "deepseek-v3-0324", "name": "DeepSeek Chat 0324", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.25, "output": 0.7 } }, "Gemma-4-31B-Musica-v1": { "id": "Gemma-4-31B-Musica-v1", "name": "Gemma 4 31B Musica v1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "mirothinker-1-7-deepresearch-mini": { "id": "mirothinker-1-7-deepresearch-mini", "name": "MiroThinker 1.7 Deep Research Mini", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 1.25, "output": 10 } }, "Baichuan4-Turbo": { "id": "Baichuan4-Turbo", "name": "Baichuan 4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 2.42, "output": 2.42 } }, "doubao-seed-1-6-flash-250615": { "id": "doubao-seed-1-6-flash-250615", "name": "Doubao Seed 1.6 Flash", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-15", "last_updated": "2025-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 16384 }, "cost": { "input": 0.0374, "output": 0.374 } }, "kimi-k2-instruct-fast": { "id": "kimi-k2-instruct-fast", "name": "Kimi K2 0711 Fast", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-15", "last_updated": "2025-07-15", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 16384 }, "cost": { "input": 0.1, "output": 2 } }, "glm-4-plus": { "id": "glm-4-plus", "name": "GLM-4 Plus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-08-01", "last_updated": "2024-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 7.497, "output": 7.497 } }, "v0-1.0-md": { "id": "v0-1.0-md", "name": "v0 1.0 MD", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-04", "last_updated": "2025-07-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "qwen3-coder-30b-a3b-instruct": { "id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3 Coder 30B A3B Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4 } }, "gemini-2.5-flash-preview-09-2025-thinking": { "id": "gemini-2.5-flash-preview-09-2025-thinking", "name": "Gemini 2.5 Flash Preview (09/2025) – Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5 } }, "v0-1.5-lg": { "id": "v0-1.5-lg", "name": "v0 1.5 LG", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-04", "last_updated": "2025-07-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 15, "output": 75 } }, "qwen3.5-flash:thinking": { "id": "qwen3.5-flash:thinking", "name": "Qwen3.5 Flash Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991808, "input": 991808, "output": 65536 }, "cost": { "input": 0.09, "output": 0.36 } }, "jamba-large": { "id": "jamba-large", "name": "Jamba Large", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 4096 }, "cost": { "input": 1.989, "output": 7.99 } }, "kimi-thinking-preview": { "id": "kimi-thinking-preview", "name": "Kimi Thinking Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 31.46, "output": 31.46 } }, "gemini-2.0-flash-thinking-exp-01-21": { "id": "gemini-2.0-flash-thinking-exp-01-21", "name": "Gemini 2.0 Flash Thinking 0121", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-01-21", "last_updated": "2025-01-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 8192 }, "cost": { "input": 0.306, "output": 1.003 } }, "claude-opus-4-1-thinking:1024": { "id": "claude-opus-4-1-thinking:1024", "name": "Claude 4.1 Opus Thinking (1K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "Qwen3.5-27B-NaNovel-Derestricted-Lite": { "id": "Qwen3.5-27B-NaNovel-Derestricted-Lite", "name": "Qwen3.5 27B NaNovel Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "doubao-seed-1-6-250615": { "id": "doubao-seed-1-6-250615", "name": "Doubao Seed 1.6", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-15", "last_updated": "2025-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 16384 }, "cost": { "input": 0.204, "output": 0.51 } }, "ernie-5.0-thinking-preview": { "id": "ernie-5.0-thinking-preview", "name": "Ernie 5.0 Thinking Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 1.1, "output": 2 } }, "gemini-2.5-flash-preview-05-20:thinking": { "id": "gemini-2.5-flash-preview-05-20:thinking", "name": "Gemini 2.5 Flash 0520 Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048000, "input": 1048000, "output": 65536 }, "cost": { "input": 0.15, "output": 3.5 } }, "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted": { "id": "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted", "name": "Qwen3.5 27B Omega Evolution v2.2 Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "azure-o1": { "id": "azure-o1", "name": "Azure o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-17", "last_updated": "2024-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 14.994, "output": 59.993 } }, "qwen3.7-plus": { "id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991808, "input": 991808, "output": 65536 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.04 } }, "glm-4-airx": { "id": "glm-4-airx", "name": "GLM-4 AirX", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-06-05", "last_updated": "2024-06-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "input": 8000, "output": 4096 }, "cost": { "input": 2.006, "output": 2.006 } }, "step-r1-v-mini": { "id": "step-r1-v-mini", "name": "Step R1 V Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-08", "last_updated": "2025-04-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 2.5, "output": 11 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2.5, "output": 10 } }, "claude-haiku-4-5-20251001-thinking": { "id": "claude-haiku-4-5-20251001-thinking", "name": "Claude Haiku 4.5 Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1 } }, "jamba-mini-1.6": { "id": "jamba-mini-1.6", "name": "Jamba Mini 1.6", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-01", "last_updated": "2025-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 4096 }, "cost": { "input": 0.1989, "output": 0.408 } }, "claude-haiku-4-5-20251001": { "id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5 } }, "claude-3-5-haiku-20241022": { "id": "claude-3-5-haiku-20241022", "name": "Claude 3.5 Haiku", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 8192 }, "cost": { "input": 0.8, "output": 4 } }, "claude-sonnet-4-thinking:32768": { "id": "claude-sonnet-4-thinking:32768", "name": "Claude 4 Sonnet Thinking (32K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "hermes-low": { "id": "hermes-low", "name": "Hermes Low", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025 } }, "claude-sonnet-4-thinking": { "id": "claude-sonnet-4-thinking", "name": "Claude 4 Sonnet Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-02-24", "last_updated": "2025-02-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "qwen3.7-max": { "id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.25 } }, "yi-medium-200k": { "id": "yi-medium-200k", "name": "Yi Medium 200k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-03-01", "last_updated": "2024-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 4096 }, "cost": { "input": 2.499, "output": 2.499 } }, "Qwen3.5-27B-BlueStar-v2-Derestricted-Lite": { "id": "Qwen3.5-27B-BlueStar-v2-Derestricted-Lite", "name": "Qwen3.5 27B BlueStar v2 Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled": { "id": "Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled", "name": "Gemma 4 31B Claude 4.6 Opus Reasoning Distilled", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306, "cache_read": 0.0306 } }, "doubao-1-5-thinking-vision-pro-250428": { "id": "doubao-1-5-thinking-vision-pro-250428", "name": "Doubao 1.5 Thinking Vision Pro", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-15", "last_updated": "2025-05-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.55, "output": 1.43 } }, "Qwen3.5-27B-BlueStar-Derestricted-Lite": { "id": "Qwen3.5-27B-BlueStar-Derestricted-Lite", "name": "Qwen3.5 27B BlueStar Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Qwen2.5-32B-EVA-v0.2": { "id": "Qwen2.5-32B-EVA-v0.2", "name": "Qwen 2.5 32b EVA", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-09-01", "last_updated": "2024-09-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 24576, "input": 24576, "output": 8192 }, "cost": { "input": 0.493, "output": 0.493 } }, "gemini-2.5-flash": { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5 } }, "qwen-long": { "id": "qwen-long", "name": "Qwen Long 10M", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-25", "last_updated": "2025-01-25", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 10000000, "input": 10000000, "output": 8192 }, "cost": { "input": 0.1003, "output": 0.408 } }, "Gemma-4-31B-DarkIdol": { "id": "Gemma-4-31B-DarkIdol", "name": "Gemma 4 31B DarkIdol", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "step-2-16k-exp": { "id": "step-2-16k-exp", "name": "Step-2 16k Exp", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-05", "last_updated": "2024-07-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16000, "input": 16000, "output": 8192 }, "cost": { "input": 7.004, "output": 19.992 } }, "Qwen3.5-27B-Marvin-V2-Derestricted-Lite": { "id": "Qwen3.5-27B-Marvin-V2-Derestricted-Lite", "name": "Qwen3.5 27B Marvin V2 Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Qwen3.5-27B-NaNovel-Derestricted": { "id": "Qwen3.5-27B-NaNovel-Derestricted", "name": "Qwen3.5 27B NaNovel Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "ernie-4.5-8k-preview": { "id": "ernie-4.5-8k-preview", "name": "Ernie 4.5 8k Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "input": 8000, "output": 16384 }, "cost": { "input": 0.66, "output": 2.6 } }, "gemini-2.0-flash-exp-image-generation": { "id": "gemini-2.0-flash-exp-image-generation", "name": "Gemini Text + Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32767, "input": 32767, "output": 8192 }, "cost": { "input": 0.2, "output": 0.8 } }, "gemini-2.5-flash-lite-preview-09-2025": { "id": "gemini-2.5-flash-lite-preview-09-2025", "name": "Gemini 2.5 Flash Lite Preview (09/2025)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4 } }, "deepseek-reasoner-cheaper": { "id": "deepseek-reasoner-cheaper", "name": "Deepseek R1 Cheaper", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.4, "output": 1.7 } }, "venice-uncensored": { "id": "venice-uncensored", "name": "Venice Uncensored", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-24", "last_updated": "2025-02-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.4, "output": 0.4 } }, "gemini-2.0-flash-001": { "id": "gemini-2.0-flash-001", "name": "Gemini 2.0 Flash", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2024-12-11", "last_updated": "2024-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 8192 }, "cost": { "input": 0.1003, "output": 0.408 } }, "gemma-4-31B-Larkspur-v0.5": { "id": "gemma-4-31B-Larkspur-v0.5", "name": "Gemma 4 31B Larkspur v0.5", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "claude-opus-4-1-20250805": { "id": "claude-opus-4-1-20250805", "name": "Claude 4.1 Opus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "holo3-35b-a3b": { "id": "holo3-35b-a3b", "name": "Holo3-35B-A3B", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 65536 }, "cost": { "input": 0.25, "output": 1.8 } }, "qwen3.5-omni-plus": { "id": "qwen3.5-omni-plus", "name": "Qwen3.5 Omni Plus", "description": "Omni-modal model for text, vision, audio, and multimodal agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-03-30", "last_updated": "2026-03-30", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983616, "input": 983616, "output": 65536 }, "cost": { "input": 0, "output": 0 } }, "Gemma-4-31B-it": { "id": "Gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-09", "last_updated": "2026-04-09", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Qwen3.5-27B-Derestricted": { "id": "Qwen3.5-27B-Derestricted", "name": "Qwen3.5 27B Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "sonar-reasoning-pro": { "id": "sonar-reasoning-pro", "name": "Perplexity Reasoning Pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "input": 127000, "output": 128000 }, "cost": { "input": 2.006, "output": 7.9985 } }, "venice-uncensored:web": { "id": "venice-uncensored:web", "name": "Venice Uncensored Web", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-05-01", "last_updated": "2024-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 80000, "input": 80000, "output": 16384 }, "cost": { "input": 0.4, "output": 0.4 } }, "doubao-1.5-pro-32k": { "id": "doubao-1.5-pro-32k", "name": "Doubao 1.5 Pro 32k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-22", "last_updated": "2025-01-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 8192 }, "cost": { "input": 0.1343, "output": 0.3349 } }, "qwen3-30b-a3b-instruct-2507": { "id": "qwen3-30b-a3b-instruct-2507", "name": "Qwen3 30B A3B Instruct 2507", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-20", "last_updated": "2025-02-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.2, "output": 0.5 } }, "ernie-5.1:thinking": { "id": "ernie-5.1:thinking", "name": "ERNIE 5.1 Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-05-10", "last_updated": "2026-05-10", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 119000, "input": 119000, "output": 64000 }, "cost": { "input": 0.75, "output": 3, "cache_read": 0.75 } }, "mirothinker-1-7-deepresearch": { "id": "mirothinker-1-7-deepresearch", "name": "MiroThinker 1.7 Deep Research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 4, "output": 25 } }, "hermes-high": { "id": "hermes-high", "name": "Hermes High", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007 } }, "jamba-large-1.6": { "id": "jamba-large-1.6", "name": "Jamba Large 1.6", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 4096 }, "cost": { "input": 1.989, "output": 7.99 } }, "glm-4-long": { "id": "glm-4-long", "name": "GLM-4 Long", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-08-01", "last_updated": "2024-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 4096 }, "cost": { "input": 0.2006, "output": 0.2006 } }, "claude-opus-4-thinking:32768": { "id": "claude-opus-4-thinking:32768", "name": "Claude 4 Opus Thinking (32K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "claude-sonnet-4-thinking:8192": { "id": "claude-sonnet-4-thinking:8192", "name": "Claude 4 Sonnet Thinking (8K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "azure-o3-mini": { "id": "azure-o3-mini", "name": "Azure o3-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 65536 }, "cost": { "input": 1.088, "output": 4.3996 } }, "qvq-max": { "id": "qvq-max", "name": "Qwen: QvQ Max", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-28", "last_updated": "2025-03-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 1.4, "output": 5.3 } }, "Qwen3.5-27B-BlueStar-v2-Derestricted": { "id": "Qwen3.5-27B-BlueStar-v2-Derestricted", "name": "Qwen3.5 27B BlueStar v2 Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Qwen3.5-27B-Vivid-Durian": { "id": "Qwen3.5-27B-Vivid-Durian", "name": "Qwen3.5 27B Vivid Durian", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "jamba-large-1.7": { "id": "jamba-large-1.7", "name": "Jamba Large 1.7", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 4096 }, "cost": { "input": 1.989, "output": 7.99 } }, "Baichuan-M2": { "id": "Baichuan-M2", "name": "Baichuan M2 32B Medical", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 15.73, "output": 15.73 } }, "Magistral-Small-2506": { "id": "Magistral-Small-2506", "name": "Magistral Small 2506", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.4, "output": 1.4 } }, "deepseek-r1": { "id": "deepseek-r1", "name": "DeepSeek R1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.4, "output": 1.7 } }, "Qwen3.5-27B-earica-Derestricted-Lite": { "id": "Qwen3.5-27B-earica-Derestricted-Lite", "name": "Qwen3.5 27B earica Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "doubao-seed-2-0-pro-260215": { "id": "doubao-seed-2-0-pro-260215", "name": "Doubao Seed 2.0 Pro", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 128000 }, "cost": { "input": 0.782, "output": 3.876 } }, "step-2-mini": { "id": "step-2-mini", "name": "Step-2 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-05", "last_updated": "2024-07-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "input": 8000, "output": 4096 }, "cost": { "input": 0.2006, "output": 0.408 } }, "gemini-exp-1206": { "id": "gemini-exp-1206", "name": "Gemini 2.0 Pro 1206", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2097152, "input": 2097152, "output": 8192 }, "cost": { "input": 1.258, "output": 4.998 } }, "claude-opus-4-5-20251101:thinking": { "id": "claude-opus-4-5-20251101:thinking", "name": "Claude 4.5 Opus Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 4.998, "output": 25.007 } }, "Qwen3.5-27B-BlueStar-v3-Derestricted": { "id": "Qwen3.5-27B-BlueStar-v3-Derestricted", "name": "Qwen3.5 27B BlueStar v3 Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "claude-opus-4-thinking": { "id": "claude-opus-4-thinking", "name": "Claude 4 Opus Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-07-15", "last_updated": "2025-07-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "qwen3.5-flash": { "id": "qwen3.5-flash", "name": "Qwen3.5 Flash", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991808, "input": 991808, "output": 65536 }, "cost": { "input": 0.09, "output": 0.36 } }, "exa-research-pro": { "id": "exa-research-pro", "name": "Exa (Research Pro)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-04", "last_updated": "2025-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 2.5, "output": 2.5 } }, "jamba-mini": { "id": "jamba-mini", "name": "Jamba Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 4096 }, "cost": { "input": 0.1989, "output": 0.408 } }, "claude-opus-4-1-thinking:32000": { "id": "claude-opus-4-1-thinking:32000", "name": "Claude 4.1 Opus Thinking (32K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "KAT-Coder-Air-V1": { "id": "KAT-Coder-Air-V1", "name": "KAT Coder Air V1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.2 } }, "exa-research": { "id": "exa-research", "name": "Exa (Research)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-04", "last_updated": "2025-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 8192 }, "cost": { "input": 2.5, "output": 2.5 } }, "deepclaude": { "id": "deepclaude", "name": "DeepClaude", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-01", "last_updated": "2025-02-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 3, "output": 15 } }, "gemini-2.5-flash-preview-09-2025": { "id": "gemini-2.5-flash-preview-09-2025", "name": "Gemini 2.5 Flash Preview (09/2025)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5 } }, "claude-opus-4-5-20251101": { "id": "claude-opus-4-5-20251101", "name": "Claude 4.5 Opus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": true, "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 4.998, "output": 25.007 } }, "GLM-4.6-Derestricted-v5": { "id": "GLM-4.6-Derestricted-v5", "name": "GLM 4.6 Derestricted v5", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 8192 }, "cost": { "input": 0.4, "output": 1.5 } }, "glm-z1-airx": { "id": "glm-z1-airx", "name": "GLM Z1 AirX", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 16384 }, "cost": { "input": 0.7, "output": 0.7 } }, "claude-sonnet-4-thinking:1024": { "id": "claude-sonnet-4-thinking:1024", "name": "Claude 4 Sonnet Thinking (1K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "holo3-35b-a3b:thinking": { "id": "holo3-35b-a3b:thinking", "name": "Holo3-35B-A3B Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 65536 }, "cost": { "input": 0.25, "output": 1.8 } }, "owl": { "id": "owl", "name": "OWL", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 262144 }, "cost": { "input": 0.1, "output": 0.3 } }, "gemini-2.0-pro-reasoner": { "id": "gemini-2.0-pro-reasoner", "name": "Gemini 2.0 Pro Reasoner", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-05", "last_updated": "2025-02-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 1.292, "output": 4.998 } }, "Gemma-4-31B-Cognitive-Unshackled": { "id": "Gemma-4-31B-Cognitive-Unshackled", "name": "Gemma 4 31B Cognitive Unshackled", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Qwen3.5-27B-earica-Derestricted": { "id": "Qwen3.5-27B-earica-Derestricted", "name": "Qwen3.5 27B earica Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "command-a-plus-05-2026": { "id": "command-a-plus-05-2026", "name": "Cohere Command A+ (05/2026)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": true, "release_date": "2026-05-22", "last_updated": "2026-05-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 64000 }, "cost": { "input": 2.5, "output": 10 } }, "auto-model-premium": { "id": "auto-model-premium", "name": "Auto model (Premium)", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 1000000 }, "cost": { "input": 9.996, "output": 19.992 } }, "learnlm-1.5-pro-experimental": { "id": "learnlm-1.5-pro-experimental", "name": "Gemini LearnLM Experimental", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-05-14", "last_updated": "2024-05-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32767, "input": 32767, "output": 8192 }, "cost": { "input": 3.502, "output": 10.506 } }, "deepseek-r1-sambanova": { "id": "deepseek-r1-sambanova", "name": "DeepSeek R1 Fast", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-20", "last_updated": "2025-02-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 4.998, "output": 6.987 } }, "claw-medium": { "id": "claw-medium", "name": "Claw Medium", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "input": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "claude-opus-4-20250514": { "id": "claude-opus-4-20250514", "name": "Claude 4 Opus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-05-14", "last_updated": "2025-05-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "yi-large": { "id": "yi-large", "name": "Yi Large", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 4096 }, "cost": { "input": 3.196, "output": 3.196 } }, "qwen3-max-2026-01-23": { "id": "qwen3-max-2026-01-23", "name": "Qwen3 Max 2026-01-23", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-01-26", "last_updated": "2026-01-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 1.2002, "output": 6.001 } }, "phi-4-mini-instruct": { "id": "phi-4-mini-instruct", "name": "Phi 4 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.17, "output": 0.68 } }, "ernie-x1-turbo-32k": { "id": "ernie-x1-turbo-32k", "name": "Ernie X1 Turbo 32k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-08", "last_updated": "2025-05-08", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 16384 }, "cost": { "input": 0.165, "output": 0.66 } }, "claw-low": { "id": "claw-low", "name": "Claw Low", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025 } }, "gemini-3-pro-image-preview": { "id": "gemini-3-pro-image-preview", "name": "Gemini 3 Pro Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2, "output": 12 } }, "gemma-4-31B-Garnet": { "id": "gemma-4-31B-Garnet", "name": "Gemma 4 31B Garnet", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "doubao-1.5-vision-pro-32k": { "id": "doubao-1.5-vision-pro-32k", "name": "Doubao 1.5 Vision Pro 32k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-22", "last_updated": "2025-01-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 8192 }, "cost": { "input": 0.459, "output": 1.377 } }, "auto-model-standard": { "id": "auto-model-standard", "name": "Auto model (Standard)", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 1000000 }, "cost": { "input": 9.996, "output": 19.992 } }, "Qwen3.5-27B-Marvin-DPO-V2-Derestricted-Lite": { "id": "Qwen3.5-27B-Marvin-DPO-V2-Derestricted-Lite", "name": "Qwen3.5 27B Marvin DPO V2 Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "glm-4": { "id": "glm-4", "name": "GLM-4", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-16", "last_updated": "2024-01-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 14.994, "output": 14.994 } }, "Qwen3.5-27B-Writer-Derestricted-Lite": { "id": "Qwen3.5-27B-Writer-Derestricted-Lite", "name": "Qwen3.5 27B Writer Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "qwen-3.6-plus": { "id": "qwen-3.6-plus", "name": "Qwen 3.6 Plus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "qwen3.6", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991800, "output": 65536 }, "cost": { "input": 0.45, "output": 2.7 } }, "brave-pro": { "id": "brave-pro", "name": "Brave (Pro)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-03-02", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 8192 }, "cost": { "input": 5, "output": 5 } }, "deepseek-chat-cheaper": { "id": "deepseek-chat-cheaper", "name": "DeepSeek V3/Chat Cheaper", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.25, "output": 0.7 } }, "Gemma-4-31B-Queen": { "id": "Gemma-4-31B-Queen", "name": "Gemma 4 31B Queen", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Gemma-4-31B-Gemopus": { "id": "Gemma-4-31B-Gemopus", "name": "Gemma 4 31B Gemopus", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "mistral-code-latest": { "id": "mistral-code-latest", "name": "Mistral Code Latest", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9 } }, "gemini-2.5-flash-lite": { "id": "gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4 } }, "jamba-mini-1.7": { "id": "jamba-mini-1.7", "name": "Jamba Mini 1.7", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 4096 }, "cost": { "input": 0.1989, "output": 0.408 } }, "universal-summarizer": { "id": "universal-summarizer", "name": "Universal Summarizer", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-05-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 30, "output": 30 } }, "Qwen3.5-27B-Anko": { "id": "Qwen3.5-27B-Anko", "name": "Qwen3.5 27B Anko", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "claude-sonnet-4-20250514": { "id": "claude-sonnet-4-20250514", "name": "Claude 4 Sonnet", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "ernie-x1-32k-preview": { "id": "ernie-x1-32k-preview", "name": "Ernie X1 32k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-03", "last_updated": "2025-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 16384 }, "cost": { "input": 0.33, "output": 1.32 } }, "azure-gpt-4-turbo": { "id": "azure-gpt-4-turbo", "name": "Azure gpt-4-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-11-06", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 9.996, "output": 30.005 } }, "Meta-Llama-3-1-8B-Instruct-FP8": { "id": "Meta-Llama-3-1-8B-Instruct-FP8", "name": "Llama 3.1 8B (decentralized)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.02, "output": 0.03 } }, "claude-sonnet-4-5-20250929-thinking": { "id": "claude-sonnet-4-5-20250929-thinking", "name": "Claude Sonnet 4.5 Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "asi1-mini": { "id": "asi1-mini", "name": "ASI1 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 1, "output": 1 } }, "qwen3.5-27b": { "id": "qwen3.5-27b", "name": "Qwen3.5 27B", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.27, "output": 2.16 } }, "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted-Lite": { "id": "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted-Lite", "name": "Qwen3.5 27B Omega Evolution v2.2 Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "claude-opus-4-1-thinking:32768": { "id": "claude-opus-4-1-thinking:32768", "name": "Claude 4.1 Opus Thinking (32K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "gemma-4-31B-K1-v5": { "id": "gemma-4-31B-K1-v5", "name": "Gemma 4 31B K1 v5", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "sonar": { "id": "sonar", "name": "Perplexity Simple", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 127000, "input": 127000, "output": 128000 }, "cost": { "input": 1.003, "output": 1.003 } }, "MiniMax-M2": { "id": "MiniMax-M2", "name": "MiniMax M2", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-10-25", "last_updated": "2025-10-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 131072 }, "cost": { "input": 0.17, "output": 1.53 } }, "sarvam-105b": { "id": "sarvam-105b", "name": "Sarvam 105B", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 4096 }, "cost": { "input": 0.045, "output": 0.177, "cache_read": 0.028 } }, "Qwen3.5-27B-Writer-V2-Derestricted-Lite": { "id": "Qwen3.5-27B-Writer-V2-Derestricted-Lite", "name": "Qwen3.5 27B Writer V2 Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Qwen3.5-27B-Marvin-V2-Derestricted": { "id": "Qwen3.5-27B-Marvin-V2-Derestricted", "name": "Qwen3.5 27B Marvin V2 Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "claw-high": { "id": "claw-high", "name": "Claw High", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007 } }, "qwen-max": { "id": "qwen-max", "name": "Qwen 2.5 Max", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-04-03", "last_updated": "2024-04-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 8192 }, "cost": { "input": 1.5997, "output": 6.392 } }, "gemma-4-31B-Fabled": { "id": "gemma-4-31B-Fabled", "name": "Gemma 4 31B Fabled", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "v0-1.5-md": { "id": "v0-1.5-md", "name": "v0 1.5 MD", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-04", "last_updated": "2025-07-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15 } }, "claude-opus-4-thinking:1024": { "id": "claude-opus-4-thinking:1024", "name": "Claude 4 Opus Thinking (1K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "gemini-2.5-flash-preview-04-17": { "id": "gemini-2.5-flash-preview-04-17", "name": "Gemini 2.5 Flash Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-04-17", "last_updated": "2025-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.15, "output": 0.6 } }, "gemini-2.5-pro-exp-03-25": { "id": "gemini-2.5-pro-exp-03-25", "name": "Gemini 2.5 Pro Experimental 0325", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2.5, "output": 10 } }, "hermes-medium": { "id": "hermes-medium", "name": "Hermes Medium", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-11", "last_updated": "2026-05-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "input": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "sonar-pro": { "id": "sonar-pro", "name": "Perplexity Pro", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 2.992, "output": 14.994 } }, "doubao-1-5-thinking-pro-250415": { "id": "doubao-1-5-thinking-pro-250415", "name": "Doubao 1.5 Thinking Pro", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-17", "last_updated": "2025-04-17", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.6, "output": 2.4 } }, "qwen3.7-max:thinking": { "id": "qwen3.7-max:thinking", "name": "Qwen3.7 Max Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 262144 } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 65536 }, "cost": { "input": 2.5, "output": 7.5, "cache_read": 0.25 } }, "ernie-4.5-turbo-vl-32k": { "id": "ernie-4.5-turbo-vl-32k", "name": "Ernie 4.5 Turbo VL 32k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-08", "last_updated": "2025-05-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 16384 }, "cost": { "input": 0.495, "output": 1.43 } }, "gemini-2.5-pro-preview-05-06": { "id": "gemini-2.5-pro-preview-05-06", "name": "Gemini 2.5 Pro Preview 0506", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-05-06", "last_updated": "2025-05-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2.5, "output": 10 } }, "hunyuan-turbos-20250226": { "id": "hunyuan-turbos-20250226", "name": "Hunyuan Turbo S", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 24000, "input": 24000, "output": 8192 }, "cost": { "input": 0.187, "output": 0.374 } }, "claude-sonnet-4-thinking:64000": { "id": "claude-sonnet-4-thinking:64000", "name": "Claude 4 Sonnet Thinking (64K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "ernie-x1-32k": { "id": "ernie-x1-32k", "name": "Ernie X1 32k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-08", "last_updated": "2025-05-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 16384 }, "cost": { "input": 0.33, "output": 1.32 } }, "command-a-reasoning-08-2025": { "id": "command-a-reasoning-08-2025", "name": "Cohere Command A (08/2025)", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-22", "last_updated": "2025-08-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 8192 }, "cost": { "input": 2.5, "output": 10 } }, "doubao-seed-2-0-code-preview-260215": { "id": "doubao-seed-2-0-code-preview-260215", "name": "Doubao Seed 2.0 Code Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 128000 }, "cost": { "input": 0.782, "output": 3.893 } }, "doubao-1.5-pro-256k": { "id": "doubao-1.5-pro-256k", "name": "Doubao 1.5 Pro 256k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 16384 }, "cost": { "input": 0.799, "output": 1.445 } }, "qwen3.5-35b-a3b": { "id": "qwen3.5-35b-a3b", "name": "Qwen3.5 35B A3B", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.225, "output": 1.8 } }, "qwen3.5-122b-a10b:thinking": { "id": "qwen3.5-122b-a10b:thinking", "name": "Qwen3.5 122B A10B Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.36, "output": 2.88 } }, "yi-lightning": { "id": "yi-lightning", "name": "Yi Lightning", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-10-16", "last_updated": "2024-10-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 12000, "input": 12000, "output": 4096 }, "cost": { "input": 0.2006, "output": 0.2006 } }, "claude-sonnet-4-5-20250929": { "id": "claude-sonnet-4-5-20250929", "name": "Claude Sonnet 4.5", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 64000 }, "cost": { "input": 2.992, "output": 14.994 } }, "deepseek-math-v2": { "id": "deepseek-math-v2", "name": "DeepSeek Math V2", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-03", "last_updated": "2025-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.6, "output": 2.2 } }, "claude-opus-4-1-thinking:8192": { "id": "claude-opus-4-1-thinking:8192", "name": "Claude 4.1 Opus Thinking (8K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "deepseek-reasoner": { "id": "deepseek-reasoner", "name": "DeepSeek Reasoner", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "input": 64000, "output": 65536 }, "cost": { "input": 0.4, "output": 1.7 } }, "gemini-2.5-flash-nothinking": { "id": "gemini-2.5-flash-nothinking", "name": "Gemini 2.5 Flash (No Thinking)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5 } }, "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted-Lite": { "id": "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted-Lite", "name": "Qwen3.5 27B Omega Evolution v2.0 Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "fastgpt": { "id": "fastgpt", "name": "Web Answer", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-08-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 7.5, "output": 7.5 } }, "Qwen3.5-27B-Infracelestial": { "id": "Qwen3.5-27B-Infracelestial", "name": "Qwen3.5 27B Infracelestial", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "glm-4-flash": { "id": "glm-4-flash", "name": "GLM-4 Flash", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-08-01", "last_updated": "2024-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 0.1003, "output": 0.1003 } }, "azure-gpt-4o-mini": { "id": "azure-gpt-4o-mini", "name": "Azure gpt-4o-mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.1496, "output": 0.595 } }, "sonar-deep-research": { "id": "sonar-deep-research", "name": "Perplexity Deep Research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-25", "last_updated": "2025-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 60000, "input": 60000, "output": 128000 }, "cost": { "input": 3.4, "output": 13.6 } }, "qwq-32b": { "id": "qwq-32b", "name": "Qwen: QwQ 32B", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 0.25599999, "output": 0.30499999 } }, "mistral-code-agent-latest": { "id": "mistral-code-agent-latest", "name": "Mistral Code Agent Latest", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 32768 }, "cost": { "input": 0.4, "output": 2 } }, "glm-4.1v-thinking-flash": { "id": "glm-4.1v-thinking-flash", "name": "GLM 4.1V Thinking Flash", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "input": 64000, "output": 8192 }, "cost": { "input": 0.3, "output": 0.3 } }, "qwen3.5-omni-flash": { "id": "qwen3.5-omni-flash", "name": "Qwen3.5 Omni Flash", "description": "Omni-modal model for text, vision, audio, and multimodal agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-03-30", "last_updated": "2026-03-30", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 49152, "input": 49152, "output": 16384 }, "cost": { "input": 0, "output": 0 } }, "gemma-4-31B-MeroMero": { "id": "gemma-4-31B-MeroMero", "name": "Gemma 4 31B MeroMero", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "gemini-2.5-flash-preview-04-17:thinking": { "id": "gemini-2.5-flash-preview-04-17:thinking", "name": "Gemini 2.5 Flash Preview Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-04-17", "last_updated": "2025-04-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.15, "output": 3.5 } }, "glm-4-air": { "id": "glm-4-air", "name": "GLM-4 Air", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-06-05", "last_updated": "2024-06-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 0.2006, "output": 0.2006 } }, "doubao-seed-1-6-thinking-250615": { "id": "doubao-seed-1-6-thinking-250615", "name": "Doubao Seed 1.6 Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-15", "last_updated": "2025-06-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 16384 }, "cost": { "input": 0.204, "output": 2.04 } }, "auto-model-basic": { "id": "auto-model-basic", "name": "Auto model (Basic)", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 1000000 }, "cost": { "input": 9.996, "output": 19.992 } }, "Qwen3.5-27B-RpRMax-v1": { "id": "Qwen3.5-27B-RpRMax-v1", "name": "Qwen3.5 27B RpRMax v1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "ernie-5.1": { "id": "ernie-5.1", "name": "ERNIE 5.1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-05-10", "last_updated": "2026-05-10", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 119000, "input": 119000, "output": 64000 }, "cost": { "input": 0.75, "output": 3, "cache_read": 0.75 } }, "Qwen3.5-27B-Marvin-DPO-V2-Derestricted": { "id": "Qwen3.5-27B-Marvin-DPO-V2-Derestricted", "name": "Qwen3.5 27B Marvin DPO V2 Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "azure-gpt-4o": { "id": "azure-gpt-4o", "name": "Azure gpt-4o", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 2.499, "output": 9.996 } }, "deepseek-chat": { "id": "deepseek-chat", "name": "DeepSeek V3/Deepseek Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.25, "output": 0.7 } }, "gemini-2.5-flash-preview-05-20": { "id": "gemini-2.5-flash-preview-05-20", "name": "Gemini 2.5 Flash 0520", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048000, "input": 1048000, "output": 65536 }, "cost": { "input": 0.15, "output": 0.6 } }, "mercury-2": { "id": "mercury-2", "name": "Mercury 2", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 50000 }, "cost": { "input": 0.25, "output": 0.75, "cache_read": 0.025 } }, "qwen3.7-plus:thinking": { "id": "qwen3.7-plus:thinking", "name": "Qwen3.7 Plus Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 262144 } ], "tool_call": false, "structured_output": false, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983616, "input": 983616, "output": 65536 }, "cost": { "input": 0.4, "output": 1.6, "cache_read": 0.04 } }, "gemini-2.0-flash-thinking-exp-1219": { "id": "gemini-2.0-flash-thinking-exp-1219", "name": "Gemini 2.0 Flash Thinking 1219", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-19", "last_updated": "2024-12-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32767, "input": 32767, "output": 8192 }, "cost": { "input": 0.1003, "output": 0.408 } }, "glm-4-plus-0111": { "id": "glm-4-plus-0111", "name": "GLM 4 Plus 0111", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-19", "last_updated": "2025-02-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 9.996, "output": 9.996 } }, "brave": { "id": "brave", "name": "Brave (Answers)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-03-02", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 8192 }, "cost": { "input": 5, "output": 5 } }, "glm-zero-preview": { "id": "glm-zero-preview", "name": "GLM Zero Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "input": 8000, "output": 4096 }, "cost": { "input": 1.802, "output": 1.802 } }, "gemini-2.5-flash-lite-preview-06-17": { "id": "gemini-2.5-flash-lite-preview-06-17", "name": "Gemini 2.5 Flash Lite Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.15, "output": 0.6 } }, "Qwen3.5-27B-Writer-Derestricted": { "id": "Qwen3.5-27B-Writer-Derestricted", "name": "Qwen3.5 27B Writer Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "KAT-Coder-Exp-72B-1010": { "id": "KAT-Coder-Exp-72B-1010", "name": "KAT Coder Exp 72B 1010", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.2 } }, "gemini-2.5-pro-preview-06-05": { "id": "gemini-2.5-pro-preview-06-05", "name": "Gemini 2.5 Pro Preview 0605", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2.5, "output": 10 } }, "Qwen3.5-27B-Musica-v1": { "id": "Qwen3.5-27B-Musica-v1", "name": "Qwen3.5 27B Musica v1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "MiniMax-M1": { "id": "MiniMax-M1", "name": "MiniMax M1", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-16", "last_updated": "2025-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 131072 }, "cost": { "input": 0.1394, "output": 1.3328 } }, "doubao-1-5-thinking-pro-vision-250415": { "id": "doubao-1-5-thinking-pro-vision-250415", "name": "Doubao 1.5 Thinking Pro Vision", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.6, "output": 2.4 } }, "qwen3.5-27b:thinking": { "id": "qwen3.5-27b:thinking", "name": "Qwen3.5 27B Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.27, "output": 2.16 } }, "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted": { "id": "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted", "name": "Qwen3.5 27B Omega Evolution v2.0 Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "Gemma-4-31B-GarnetV2": { "id": "Gemma-4-31B-GarnetV2", "name": "Gemma 4 31B Garnet V2", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2026-05-01", "last_updated": "2026-05-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "qwen-turbo": { "id": "qwen-turbo", "name": "Qwen Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-11-01", "last_updated": "2024-11-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 8192 }, "cost": { "input": 0.04998, "output": 0.2006 } }, "phi-4-multimodal-instruct": { "id": "phi-4-multimodal-instruct", "name": "Phi 4 Multimodal", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.07, "output": 0.11 } }, "mistral-small-31-24b-instruct": { "id": "mistral-small-31-24b-instruct", "name": "Mistral Small 31 24b Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 131072 }, "cost": { "input": 0.1, "output": 0.3 } }, "ernie-4.5-turbo-128k": { "id": "ernie-4.5-turbo-128k", "name": "Ernie 4.5 Turbo 128k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-08", "last_updated": "2025-05-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.132, "output": 0.55 } }, "Qwen3.5-27B-BlueStar-Derestricted": { "id": "Qwen3.5-27B-BlueStar-Derestricted", "name": "Qwen3.5 27B BlueStar Derestricted", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-06", "last_updated": "2026-04-06", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "gemini-2.5-flash-lite-preview-09-2025-thinking": { "id": "gemini-2.5-flash-lite-preview-09-2025-thinking", "name": "Gemini 2.5 Flash Lite Preview (09/2025) – Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4 } }, "doubao-seed-1-8-251215": { "id": "doubao-seed-1-8-251215", "name": "Doubao Seed 1.8", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.612, "output": 6.12 } }, "qwen3.6-max-preview": { "id": "qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "qwen3.6", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-04-20", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 245800, "output": 65536 }, "cost": { "input": 1.3, "output": 7.8 } }, "exa-answer": { "id": "exa-answer", "name": "Exa (Answer)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-04", "last_updated": "2025-06-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4096, "input": 4096, "output": 4096 }, "cost": { "input": 2.5, "output": 2.5 } }, "Baichuan4-Air": { "id": "Baichuan4-Air", "name": "Baichuan 4 Air", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.157, "output": 0.157 } }, "qwen3.5-122b-a10b": { "id": "qwen3.5-122b-a10b", "name": "Qwen3.5 122B A10B", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.36, "output": 2.88 } }, "sarvam-30b": { "id": "sarvam-30b", "name": "Sarvam 30B", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 4096 }, "cost": { "input": 0.028, "output": 0.111, "cache_read": 0.017 } }, "claude-opus-4-1-thinking": { "id": "claude-opus-4-1-thinking", "name": "Claude 4.1 Opus Thinking", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "claude-opus-4-thinking:32000": { "id": "claude-opus-4-thinking:32000", "name": "Claude 4 Opus Thinking (32K)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 32000 }, "cost": { "input": 14.994, "output": 75.004 } }, "auto-model": { "id": "auto-model", "name": "Auto model", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-06-01", "last_updated": "2024-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 1000000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-vl-235b-a22b-instruct-original": { "id": "qwen3-vl-235b-a22b-instruct-original", "name": "Qwen3 VL 235B A22B Instruct Original", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.5, "output": 1.2 } }, "glm-z1-air": { "id": "glm-z1-air", "name": "GLM Z1 Air", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 16384 }, "cost": { "input": 0.07, "output": 0.07 } }, "Qwen3.5-27B-Queen-Derestricted-Lite": { "id": "Qwen3.5-27B-Queen-Derestricted-Lite", "name": "Qwen3.5 27B Queen Derestricted Lite", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.306, "output": 0.306 } }, "inclusionai/ling-2.6-1t": { "id": "inclusionai/ling-2.6-1t", "name": "Ling 2.6 1T", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 32768 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.06 } }, "inclusionai/ring-2.6-1t": { "id": "inclusionai/ring-2.6-1t", "name": "Ring 2.6 1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-08", "last_updated": "2026-05-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 65536 }, "cost": { "input": 1, "output": 3 } }, "inclusionai/ling-2.6-flash": { "id": "inclusionai/ling-2.6-flash", "name": "Ling 2.6 Flash", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 32768 }, "cost": { "input": 0.08, "output": 0.24 } }, "Alibaba-NLP/Tongyi-DeepResearch-30B-A3B": { "id": "Alibaba-NLP/Tongyi-DeepResearch-30B-A3B", "name": "Tongyi DeepResearch 30B A3B", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "yi", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.08, "output": 0.24000000000000002 } }, "ibm-granite/granite-4.1-8b": { "id": "ibm-granite/granite-4.1-8b", "name": "Granite 4.1 8B", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 131072 }, "cost": { "input": 0.05, "output": 0.1, "cache_read": 0.05 } }, "Salesforce/Llama-xLAM-2-70b-fc-r": { "id": "Salesforce/Llama-xLAM-2-70b-fc-r", "name": "Llama-xLAM-2 70B fc-r", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-13", "last_updated": "2025-04-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 2.5 } }, "THUDM/GLM-Z1-32B-0414": { "id": "THUDM/GLM-Z1-32B-0414", "name": "GLM Z1 32B 0414", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm-z", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.2, "output": 0.2 } }, "THUDM/GLM-4-32B-0414": { "id": "THUDM/GLM-4-32B-0414", "name": "GLM 4 32B 0414", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.2, "output": 0.2 } }, "THUDM/GLM-4-9B-0414": { "id": "THUDM/GLM-4-9B-0414", "name": "GLM 4 9B 0414", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 8000 }, "cost": { "input": 0.2, "output": 0.2 } }, "THUDM/GLM-Z1-9B-0414": { "id": "THUDM/GLM-Z1-9B-0414", "name": "GLM Z1 9B 0414", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm-z", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 8000 }, "cost": { "input": 0.2, "output": 0.2 } }, "meta-llama/llama-3.1-8b-instruct": { "id": "meta-llama/llama-3.1-8b-instruct", "name": "Llama 3.1 8b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 16384 }, "cost": { "input": 0.0544, "output": 0.0544 } }, "meta-llama/llama-4-maverick": { "id": "meta-llama/llama-4-maverick", "name": "Llama 4 Maverick", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 65536 }, "cost": { "input": 0.18000000000000002, "output": 0.8 } }, "meta-llama/llama-3.3-70b-instruct": { "id": "meta-llama/llama-3.3-70b-instruct", "name": "Llama 3.3 70b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 16384 }, "cost": { "input": 0.05, "output": 0.23 } }, "meta-llama/llama-4-scout": { "id": "meta-llama/llama-4-scout", "name": "Llama 4 Scout", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 328000, "input": 328000, "output": 65536 }, "cost": { "input": 0.085, "output": 0.46 } }, "meta-llama/llama-3.2-3b-instruct": { "id": "meta-llama/llama-3.2-3b-instruct", "name": "Llama 3.2 3b Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 8192 }, "cost": { "input": 0.0306, "output": 0.0493 } }, "featherless-ai/Qwerky-72B": { "id": "featherless-ai/Qwerky-72B", "name": "Qwerky 72B", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "qwerky", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-20", "last_updated": "2025-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 8192 }, "cost": { "input": 0.5, "output": 0.5 } }, "moonshotai/kimi-k2-instruct-0711": { "id": "moonshotai/kimi-k2-instruct-0711", "name": "Kimi K2 0711", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-07-11", "last_updated": "2025-07-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.1, "output": 2 } }, "moonshotai/Kimi-K2-Instruct-0905": { "id": "moonshotai/Kimi-K2-Instruct-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 262144 }, "cost": { "input": 0.4, "output": 2 } }, "moonshotai/kimi-k2-thinking-original": { "id": "moonshotai/kimi-k2-thinking-original", "name": "Kimi K2 Thinking Original", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 16384 }, "cost": { "input": 0.6, "output": 2.5 } }, "moonshotai/kimi-k2.5:thinking": { "id": "moonshotai/kimi-k2.5:thinking", "name": "Kimi K2.5 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "release_date": "2026-01-26", "last_updated": "2026-01-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.9 } }, "moonshotai/kimi-k2-instruct": { "id": "moonshotai/kimi-k2-instruct", "name": "Kimi K2 Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 8192 }, "cost": { "input": 0.1, "output": 2 } }, "moonshotai/kimi-k2-thinking": { "id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 262144 }, "cost": { "input": 0.3, "output": 1.2 } }, "moonshotai/kimi-k2.6:thinking": { "id": "moonshotai/kimi-k2.6:thinking", "name": "Kimi K2.6 Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "release_date": "2026-04-16", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.53, "output": 2.73 } }, "moonshotai/kimi-k2.5": { "id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "release_date": "2026-01-26", "last_updated": "2026-01-26", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.9 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "release_date": "2026-04-16", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 65536 }, "cost": { "input": 0.53, "output": 2.73 } }, "moonshotai/kimi-latest": { "id": "moonshotai/kimi-latest", "name": "Kimi Latest", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "release_date": "2026-05-03", "last_updated": "2026-05-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 65536 }, "cost": { "input": 0.5, "output": 2.6, "cache_read": 0.125 } }, "moonshotai/kimi-k2-thinking-turbo-original": { "id": "moonshotai/kimi-k2-thinking-turbo-original", "name": "Kimi K2 Thinking Turbo Original", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 16384 }, "cost": { "input": 1.15, "output": 8 } }, "baidu/ernie-4.5-vl-28b-a3b": { "id": "baidu/ernie-4.5-vl-28b-a3b", "name": "ERNIE 4.5 VL 28B", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "ernie", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 16384 }, "cost": { "input": 0.13999999999999999, "output": 0.5599999999999999 } }, "perceptron/perceptron-mk1": { "id": "perceptron/perceptron-mk1", "name": "Perceptron Mk1", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.15, "output": 1.5 } }, "failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5": { "id": "failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5", "name": "Llama 3 70B abliterated", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 8192 }, "cost": { "input": 0.7, "output": 0.7 } }, "nanogpt/coding-router:max": { "id": "nanogpt/coding-router:max", "name": "Coding Router Max", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "nanogpt/coding-router:high": { "id": "nanogpt/coding-router:high", "name": "Coding Router High", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 1.1, "output": 2.2, "cache_read": 0.11 } }, "nanogpt/coding-router:low": { "id": "nanogpt/coding-router:low", "name": "Coding Router Low", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "nanogpt/coding-router:medium": { "id": "nanogpt/coding-router:medium", "name": "Coding Router Medium", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "nanogpt/coding-router": { "id": "nanogpt/coding-router", "name": "Coding Router", "description": "Automatic model router for matching prompts to suitable backends and budgets", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 1.1, "output": 2.2, "cache_read": 0.11 } }, "GalrionSoftworks/MN-LooseCannon-12B-v1": { "id": "GalrionSoftworks/MN-LooseCannon-12B-v1", "name": "MN-LooseCannon-12B-v1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "TheDrummer/Cydonia-24B-v4.3": { "id": "TheDrummer/Cydonia-24B-v4.3", "name": "The Drummer Cydonia 24B v4.3", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-25", "last_updated": "2025-12-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.1003, "output": 0.1207 } }, "TheDrummer/Cydonia-24B-v4.1": { "id": "TheDrummer/Cydonia-24B-v4.1", "name": "The Drummer Cydonia 24B v4.1", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 32768 }, "cost": { "input": 0.1003, "output": 0.1207 } }, "TheDrummer/Cydonia-24B-v2": { "id": "TheDrummer/Cydonia-24B-v2", "name": "The Drummer Cydonia 24B v2", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-17", "last_updated": "2025-02-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 32768 }, "cost": { "input": 0.1003, "output": 0.1207 } }, "TheDrummer/Anubis-70B-v1.1": { "id": "TheDrummer/Anubis-70B-v1.1", "name": "Anubis 70B v1.1", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 16384 }, "cost": { "input": 0.31, "output": 0.31 } }, "TheDrummer/UnslopNemo-12B-v4.1": { "id": "TheDrummer/UnslopNemo-12B-v4.1", "name": "UnslopNemo 12b v4", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.493, "output": 0.493 } }, "TheDrummer/skyfall-36b-v2": { "id": "TheDrummer/skyfall-36b-v2", "name": "TheDrummer Skyfall 36B V2", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-10", "last_updated": "2025-03-10", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "input": 64000, "output": 32768 }, "cost": { "input": 0.493, "output": 0.493 } }, "TheDrummer/Rocinante-12B-v1.1": { "id": "TheDrummer/Rocinante-12B-v1.1", "name": "Rocinante 12b", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.408, "output": 0.595 } }, "TheDrummer/Magidonia-24B-v4.3": { "id": "TheDrummer/Magidonia-24B-v4.3", "name": "The Drummer Magidonia 24B v4.3", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-25", "last_updated": "2025-12-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.1003, "output": 0.1207 } }, "TheDrummer/Cydonia-24B-v4": { "id": "TheDrummer/Cydonia-24B-v4", "name": "The Drummer Cydonia 24B v4", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 32768 }, "cost": { "input": 0.2006, "output": 0.2414 } }, "TheDrummer/Skyfall-31B-v4.2": { "id": "TheDrummer/Skyfall-31B-v4.2", "name": "TheDrummer Skyfall 31B v4.2", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-03-26", "last_updated": "2026-03-26", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 16384 }, "cost": { "input": 0.55, "output": 0.8 } }, "TheDrummer/Anubis-70B-v1": { "id": "TheDrummer/Anubis-70B-v1", "name": "Anubis 70B v1", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 16384 }, "cost": { "input": 0.31, "output": 0.31 } }, "huihui-ai/Qwen2.5-32B-Instruct-abliterated": { "id": "huihui-ai/Qwen2.5-32B-Instruct-abliterated", "name": "Qwen 2.5 32B Abliterated", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-06", "last_updated": "2025-01-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.7, "output": 0.7 } }, "huihui-ai/Llama-3.3-70B-Instruct-abliterated": { "id": "huihui-ai/Llama-3.3-70B-Instruct-abliterated", "name": "Llama 3.3 70B Instruct abliterated", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.7, "output": 0.7 } }, "huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated": { "id": "huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated", "name": "DeepSeek R1 Qwen Abliterated", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 1.4, "output": 1.4 } }, "huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated": { "id": "huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated", "name": "DeepSeek R1 Llama 70B Abliterated", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.7, "output": 0.7 } }, "Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B": { "id": "Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B", "name": "Llama 3.05 Storybreaker Ministral 70b", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B": { "id": "Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B", "name": "Nemotron Tenyxchat Storybreaker 70b", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "stepfun-ai/step-3.5-flash-2603": { "id": "stepfun-ai/step-3.5-flash-2603", "name": "Step 3.5 Flash 2603", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-14", "last_updated": "2026-04-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.1, "output": 0.3 } }, "stepfun-ai/step-3.5-flash": { "id": "stepfun-ai/step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "family": "step", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2026-02-02", "last_updated": "2026-02-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 0.5 } }, "mistral/mistral-medium-3.5": { "id": "mistral/mistral-medium-3.5", "name": "Mistral Medium 3.5", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 1.5, "output": 7.5 } }, "mistral/mistral-medium-3.5:thinking": { "id": "mistral/mistral-medium-3.5:thinking", "name": "Mistral Medium 3.5 Thinking", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 1.5, "output": 7.5 } }, "Tongyi-Zhiwen/QwenLong-L1-32B": { "id": "Tongyi-Zhiwen/QwenLong-L1-32B", "name": "QwenLong L1 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-25", "last_updated": "2025-01-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 40960 }, "cost": { "input": 0.13999999999999999, "output": 0.6 } }, "google/gemini-flash-1.5": { "id": "google/gemini-flash-1.5", "name": "Gemini 1.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-05-14", "last_updated": "2024-05-14", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "input": 2000000, "output": 8192 }, "cost": { "input": 0.0748, "output": 0.306 } }, "google/gemini-3.1-pro-preview-high": { "id": "google/gemini-3.1-pro-preview-high", "name": "Gemini 3.1 Pro (Preview High)", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-21", "last_updated": "2026-02-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025 } }, "google/gemini-3-flash-preview-thinking": { "id": "google/gemini-3-flash-preview-thinking", "name": "Gemini 3 Flash Thinking", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.5, "output": 3 } }, "google/gemini-3.5-flash": { "id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 131072 }, "cost": { "input": 0.1, "output": 0.35 } }, "google/gemini-pro-latest": { "id": "google/gemini-pro-latest", "name": "Gemini Pro Latest", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-29", "last_updated": "2026-03-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/gemini-3.1-pro-preview-customtools": { "id": "google/gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro (Preview Custom Tools)", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-27", "last_updated": "2026-02-27", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/gemini-flash-lite-latest": { "id": "google/gemini-flash-lite-latest", "name": "Gemini Flash Lite Latest", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-29", "last_updated": "2026-03-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025 } }, "google/gemma-4-26b-a4b-it:thinking": { "id": "google/gemma-4-26b-a4b-it:thinking", "name": "Gemma 4 26B A4B Thinking", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 131072 }, "cost": { "input": 0.13, "output": 0.4 } }, "google/gemini-3.1-pro-preview": { "id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro (Preview)", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/gemma-4-26b-a4b-it": { "id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 131072 }, "cost": { "input": 0.13, "output": 0.4 } }, "google/gemini-3.1-pro-preview-low": { "id": "google/gemini-3.1-pro-preview-low", "name": "Gemini 3.1 Pro (Preview Low)", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-21", "last_updated": "2026-02-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2 } }, "google/gemma-4-31b-it:thinking": { "id": "google/gemma-4-31b-it:thinking", "name": "Gemma 4 31B Thinking", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 131072 }, "cost": { "input": 0.1, "output": 0.35 } }, "google/gemini-3.5-flash-thinking": { "id": "google/gemini-3.5-flash-thinking", "name": "Gemini 3.5 Flash Thinking", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 } }, "google/gemini-3-flash-preview": { "id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash (Preview)", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 0.5, "output": 3 } }, "google/gemini-flash-latest": { "id": "google/gemini-flash-latest", "name": "Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2026-03-29", "last_updated": "2026-03-29", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048756, "input": 1048756, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15 } }, "liquid/lfm-2-24b-a2b": { "id": "liquid/lfm-2-24b-a2b", "name": "LFM2 24B A2B", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-20", "last_updated": "2025-12-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.03, "output": 0.12 } }, "x-ai/grok-4.20": { "id": "x-ai/grok-4.20", "name": "Grok 4.20", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "input": 2000000, "output": 131072 }, "cost": { "input": 2, "output": 6 } }, "x-ai/grok-4.3": { "id": "x-ai/grok-4.3", "name": "Grok 4.3", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 1000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "x-ai/grok-4.20-multi-agent": { "id": "x-ai/grok-4.20-multi-agent", "name": "Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 2000000, "input": 2000000, "output": 131072 }, "cost": { "input": 2, "output": 6 } }, "x-ai/grok-latest": { "id": "x-ai/grok-latest", "name": "Grok Latest", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-03", "last_updated": "2026-05-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 1000000 }, "cost": { "input": 1.25, "output": 2.5, "cache_read": 0.2 } }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", "description": "Grok coding model for agentic engineering, edits, and codebase workflows", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-20", "last_updated": "2026-05-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 1, "output": 2, "cache_read": 0.2 } }, "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0": { "id": "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0", "name": "EVA Llama 3.33 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 2.006, "output": 2.006 } }, "EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2": { "id": "EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2", "name": "EVA-Qwen2.5-72B-v0.2", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.7989999999999999, "output": 0.7989999999999999 } }, "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1": { "id": "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1", "name": "EVA-LLaMA-3.33-70B-v0.1", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 2.006, "output": 2.006 } }, "EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2": { "id": "EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2", "name": "EVA-Qwen2.5-32B-v0.2", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.7989999999999999, "output": 0.7989999999999999 } }, "microsoft/wizardlm-2-8x22b": { "id": "microsoft/wizardlm-2-8x22b", "name": "WizardLM-2 8x22B", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Doctor-Shotgun/MS3.2-24B-Magnum-Diamond": { "id": "Doctor-Shotgun/MS3.2-24B-Magnum-Diamond", "name": "MS3.2 24B Magnum Diamond", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 32768 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "poolside/laguna-xs.2": { "id": "poolside/laguna-xs.2", "name": "Laguna XS.2", "description": "Agentic coding model from Poolside in the XS size class for local deployment", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.2, "output": 0.4 } }, "poolside/laguna-m.1": { "id": "poolside/laguna-m.1", "name": "Laguna M.1", "description": "Poolside's flagship agentic coding model for long-horizon work", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 32768 }, "cost": { "input": 0.2, "output": 0.4 } }, "z-ai/glm-4.5v:thinking": { "id": "z-ai/glm-4.5v:thinking", "name": "GLM 4.5V Thinking", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glmv", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-11-22", "last_updated": "2025-11-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "input": 64000, "output": 96000 }, "cost": { "input": 0.6, "output": 1.7999999999999998 } }, "z-ai/glm-4.6:thinking": { "id": "z-ai/glm-4.6:thinking", "name": "GLM 4.6 Thinking", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 65535 }, "cost": { "input": 0.4, "output": 1.5 } }, "z-ai/glm-4.5v": { "id": "z-ai/glm-4.5v", "name": "GLM 4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glmv", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2025-11-22", "last_updated": "2025-11-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 64000, "input": 64000, "output": 96000 }, "cost": { "input": 0.6, "output": 1.7999999999999998 } }, "z-ai/glm-4.6": { "id": "z-ai/glm-4.6", "name": "GLM 4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 65535 }, "cost": { "input": 0.4, "output": 1.5 } }, "z-ai/glm-5v-turbo:thinking": { "id": "z-ai/glm-5v-turbo:thinking", "name": "GLM 5V Turbo Thinking", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202800, "input": 202800, "output": 131100 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "z-ai/glm-5v-turbo": { "id": "z-ai/glm-5v-turbo", "name": "GLM 5V Turbo", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202800, "input": 202800, "output": 131100 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "z-ai/glm-5-turbo": { "id": "z-ai/glm-5-turbo", "name": "GLM 5 Turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-03-15", "last_updated": "2026-03-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202800, "input": 202800, "output": 131072 }, "cost": { "input": 1.2, "output": 4, "cache_read": 0.24 } }, "openai/o3-mini-low": { "id": "openai/o3-mini-low", "name": "OpenAI o3-mini (Low)", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 9.996, "output": 19.992 } }, "openai/gpt-oss-safeguard-20b": { "id": "openai/gpt-oss-safeguard-20b", "name": "GPT OSS Safeguard 20B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-10-29", "last_updated": "2025-10-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.075, "output": 0.3 } }, "openai/o3": { "id": "openai/o3", "name": "OpenAI o3", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 2, "output": 8 } }, "openai/o4-mini-high": { "id": "openai/o4-mini-high", "name": "OpenAI o4-mini high", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4 } }, "openai/o3-pro-2025-06-10": { "id": "openai/o3-pro-2025-06-10", "name": "OpenAI o3-pro (2025-06-10)", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 9.996, "output": 19.992 } }, "openai/gpt-5.2-pro": { "id": "openai/gpt-5.2-pro", "name": "GPT 5.2 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-01-01", "last_updated": "2026-01-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 21, "output": 168 } }, "openai/gpt-4o-mini-search-preview": { "id": "openai/gpt-4o-mini-search-preview", "name": "GPT-4o mini Search Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.088, "output": 0.35 } }, "openai/gpt-5": { "id": "openai/gpt-5", "name": "GPT 5", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "openai/gpt-3.5-turbo": { "id": "openai/gpt-3.5-turbo", "name": "GPT-3.5 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2022-11-30", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16385, "input": 16385, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5 } }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", "name": "GPT 5 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 15, "output": 120 } }, "openai/gpt-4o": { "id": "openai/gpt-4o", "name": "GPT-4o", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 2.499, "output": 9.996 } }, "openai/o4-mini": { "id": "openai/o4-mini", "name": "OpenAI o4-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4 } }, "openai/gpt-5.4-nano": { "id": "openai/gpt-5.4-nano", "name": "GPT 5.4 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "openai/gpt-5.1-codex": { "id": "openai/gpt-5.1-codex", "name": "GPT 5.1 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "openai/gpt-5.1-codex-max": { "id": "openai/gpt-5.1-codex-max", "name": "GPT 5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 2.5, "output": 20 } }, "openai/gpt-4o-2024-08-06": { "id": "openai/gpt-4o-2024-08-06", "name": "GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-08-06", "last_updated": "2024-08-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 2.499, "output": 9.996 } }, "openai/o1-preview": { "id": "openai/o1-preview", "name": "OpenAI o1-preview", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2024-09-12", "last_updated": "2024-09-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 14.993999999999998, "output": 59.993 } }, "openai/o3-mini": { "id": "openai/o3-mini", "name": "OpenAI o3-mini", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 1.1, "output": 4.4 } }, "openai/gpt-5.2": { "id": "openai/gpt-5.2", "name": "GPT 5.2", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-01-01", "last_updated": "2026-01-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "openai/gpt-5.3-codex": { "id": "openai/gpt-5.3-codex", "name": "GPT 5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-24", "last_updated": "2026-02-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "openai/gpt-latest": { "id": "openai/gpt-latest", "name": "GPT Latest", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-29", "last_updated": "2026-03-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "openai/gpt-5.1-codex-mini": { "id": "openai/gpt-5.1-codex-mini", "name": "GPT 5.1 Codex Mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2 } }, "openai/o4-mini-deep-research": { "id": "openai/o4-mini-deep-research", "name": "OpenAI o4-mini Deep Research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 9.996, "output": 19.992 } }, "openai/gpt-4.1-nano": { "id": "openai/gpt-4.1-nano", "name": "GPT 4.1 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "input": 1047576, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4 } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.05, "output": 0.25 } }, "openai/gpt-4o-2024-11-20": { "id": "openai/gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 2.5, "output": 10 } }, "openai/o1": { "id": "openai/o1", "name": "OpenAI o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2024-12-17", "last_updated": "2024-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 14.993999999999998, "output": 59.993 } }, "openai/o1-pro": { "id": "openai/o1-pro", "name": "OpenAI o1 Pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-25", "last_updated": "2025-01-25", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 150, "output": 600 } }, "openai/gpt-chat-latest": { "id": "openai/gpt-chat-latest", "name": "GPT Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-05-03", "last_updated": "2026-05-03", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "openai/gpt-5.4": { "id": "openai/gpt-5.4", "name": "GPT 5.4", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 922000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25 } }, "openai/gpt-5.4-mini": { "id": "openai/gpt-5.4-mini", "name": "GPT 5.4 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "openai/gpt-4.1": { "id": "openai/gpt-4.1", "name": "GPT 4.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-10", "last_updated": "2025-09-10", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "input": 1047576, "output": 32768 }, "cost": { "input": 2, "output": 8 } }, "openai/o3-deep-research": { "id": "openai/o3-deep-research", "name": "OpenAI o3 Deep Research", "description": "Research model for long-horizon investigation, synthesis, and analytical reports", "family": "o", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["medium"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 9.996, "output": 19.992 } }, "openai/gpt-4-turbo-preview": { "id": "openai/gpt-4-turbo-preview", "name": "GPT-4 Turbo Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-11-06", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 9.996, "output": 30.004999999999995 } }, "openai/gpt-5-mini": { "id": "openai/gpt-5-mini", "name": "GPT 5 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 0.25, "output": 2 } }, "openai/gpt-4.1-mini": { "id": "openai/gpt-4.1-mini", "name": "GPT 4.1 Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1047576, "input": 1047576, "output": 32768 }, "cost": { "input": 0.4, "output": 1.6 } }, "openai/gpt-4-turbo": { "id": "openai/gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2023-11-06", "last_updated": "2024-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 10, "output": 30 } }, "openai/gpt-5-nano": { "id": "openai/gpt-5-nano", "name": "GPT 5 Nano", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 0.05, "output": 0.4 } }, "openai/gpt-5.4-pro": { "id": "openai/gpt-5.4-pro", "name": "GPT 5.4 Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 922000, "input": 922000, "output": 128000 }, "cost": { "input": 30, "output": 180, "cache_read": 3 } }, "openai/o3-mini-high": { "id": "openai/o3-mini-high", "name": "OpenAI o3-mini (High)", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-01-31", "last_updated": "2025-01-31", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 100000 }, "cost": { "input": 0.64, "output": 2.588 } }, "openai/gpt-4o-search-preview": { "id": "openai/gpt-4o-search-preview", "name": "GPT-4o Search Preview", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 1.47, "output": 5.88 } }, "openai/gpt-5.1-2025-11-13": { "id": "openai/gpt-5.1-2025-11-13", "name": "GPT-5.1 (2025-11-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 32768 }, "cost": { "input": 1.25, "output": 10 } }, "openai/gpt-4o-mini": { "id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.1496, "output": 0.595 } }, "openai/gpt-oss-20b": { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": false, "structured_output": false, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.04, "output": 0.15 } }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", "name": "GPT-5 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 9.996, "output": 19.992 } }, "openai/gpt-5.2-codex": { "id": "openai/gpt-5.2-codex", "name": "GPT 5.2 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-01-14", "last_updated": "2026-01-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 1.75, "output": 14 } }, "openai/gpt-5.1": { "id": "openai/gpt-5.1", "name": "GPT 5.1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 400000, "output": 128000 }, "cost": { "input": 1.25, "output": 10 } }, "openai/gpt-5.5": { "id": "openai/gpt-5.5", "name": "GPT 5.5", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5 } }, "Gryphe/MythoMax-L2-13b": { "id": "Gryphe/MythoMax-L2-13b", "name": "MythoMax 13B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 4000, "input": 4000, "output": 4096 }, "cost": { "input": 0.1003, "output": 0.1003 } }, "Unbabel/M-Prometheus-14B": { "id": "Unbabel/M-Prometheus-14B", "name": "M-Prometheus 14B", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.2, "output": 0.2 } }, "LLM360/K2-Think": { "id": "LLM360/K2-Think", "name": "K2-Think", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 0.17, "output": 0.68 } }, "NousResearch/hermes-4-405b": { "id": "NousResearch/hermes-4-405b", "name": "Hermes 4 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.3, "output": 1.2 } }, "NousResearch/hermes-3-llama-3.1-70b": { "id": "NousResearch/hermes-3-llama-3.1-70b", "name": "Hermes 3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-01-07", "last_updated": "2026-01-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 8192 }, "cost": { "input": 0.408, "output": 0.408 } }, "NousResearch/Hermes-4-70B:thinking": { "id": "NousResearch/Hermes-4-70B:thinking", "name": "Hermes 4 (Thinking)", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-17", "last_updated": "2025-09-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.2006, "output": 0.3995 } }, "NousResearch/hermes-4-70b": { "id": "NousResearch/hermes-4-70b", "name": "Hermes 4 Medium", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-03", "last_updated": "2025-07-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.2006, "output": 0.3995 } }, "NousResearch/DeepHermes-3-Mistral-24B-Preview": { "id": "NousResearch/DeepHermes-3-Mistral-24B-Preview", "name": "DeepHermes-3 Mistral 24B (Preview)", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-10", "last_updated": "2025-05-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.3 } }, "NousResearch/hermes-4-405b:thinking": { "id": "NousResearch/hermes-4-405b:thinking", "name": "Hermes 4 Large (Thinking)", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.3, "output": 1.2 } }, "unsloth/gemma-3-12b-it": { "id": "unsloth/gemma-3-12b-it", "name": "Gemma 3 12B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "unsloth", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-10", "last_updated": "2025-03-10", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 131072 }, "cost": { "input": 0.272, "output": 0.272 } }, "unsloth/gemma-3-4b-it": { "id": "unsloth/gemma-3-4b-it", "name": "Gemma 3 4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "unsloth", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-10", "last_updated": "2025-03-10", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.2006, "output": 0.2006 } }, "unsloth/gemma-3-27b-it": { "id": "unsloth/gemma-3-27b-it", "name": "Gemma 3 27B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "unsloth", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-10", "last_updated": "2025-03-10", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 96000 }, "cost": { "input": 0.2992, "output": 0.2992 } }, "NeverSleep/Lumimaid-v0.2-70B": { "id": "NeverSleep/Lumimaid-v0.2-70B", "name": "Lumimaid v0.2", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 1, "output": 1.5 } }, "mistralai/mixtral-8x7b-instruct-v0.1": { "id": "mistralai/mixtral-8x7b-instruct-v0.1", "name": "Mixtral 8x7B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mixtral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.27, "output": 0.27 } }, "mistralai/mistral-small-4-119b-2603:thinking": { "id": "mistralai/mistral-small-4-119b-2603:thinking", "name": "Mistral Small 4 119B Thinking", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.4, "output": 1.4 } }, "mistralai/mixtral-8x22b-instruct-v0.1": { "id": "mistralai/mixtral-8x22b-instruct-v0.1", "name": "Mixtral 8x22B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mixtral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 32768 }, "cost": { "input": 0.8999999999999999, "output": 0.8999999999999999 } }, "mistralai/Devstral-Small-2505": { "id": "mistralai/Devstral-Small-2505", "name": "Mistral Devstral Small 2505", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-02", "last_updated": "2025-08-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.060000000000000005, "output": 0.060000000000000005 } }, "mistralai/ministral-8b-2512": { "id": "mistralai/ministral-8b-2512", "name": "Ministral 8B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-04", "last_updated": "2025-12-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 32768 }, "cost": { "input": 0.15, "output": 0.15 } }, "mistralai/mistral-saba": { "id": "mistralai/mistral-saba", "name": "Mistral Saba", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-17", "last_updated": "2025-02-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 32768 }, "cost": { "input": 0.1989, "output": 0.595 } }, "mistralai/mistral-large": { "id": "mistralai/mistral-large", "name": "Mistral Large 2411", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-02-26", "last_updated": "2024-02-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 256000 }, "cost": { "input": 2.006, "output": 6.001 } }, "mistralai/mistral-medium-3.1": { "id": "mistralai/mistral-medium-3.1", "name": "Mistral Medium 3.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 32768 }, "cost": { "input": 0.4, "output": 2 } }, "mistralai/ministral-3b-2512": { "id": "mistralai/ministral-3b-2512", "name": "Ministral 3B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-04", "last_updated": "2025-12-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 32768 }, "cost": { "input": 0.1, "output": 0.1 } }, "mistralai/ministral-14b-instruct-2512": { "id": "mistralai/ministral-14b-instruct-2512", "name": "Ministral 3 14B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 32768 }, "cost": { "input": 0.1, "output": 0.4 } }, "mistralai/mistral-small-4-119b-2603": { "id": "mistralai/mistral-small-4-119b-2603", "name": "Mistral Small 4 119B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.4, "output": 1.4 } }, "mistralai/ministral-14b-2512": { "id": "mistralai/ministral-14b-2512", "name": "Ministral 14B", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-04", "last_updated": "2025-12-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 32768 }, "cost": { "input": 0.2, "output": 0.2 } }, "mistralai/mistral-large-3-675b-instruct-2512": { "id": "mistralai/mistral-large-3-675b-instruct-2512", "name": "Mistral Large 3 675B", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 256000 }, "cost": { "input": 1, "output": 3 } }, "mistralai/mistral-medium-3": { "id": "mistralai/mistral-medium-3", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 32768 }, "cost": { "input": 0.4, "output": 2 } }, "mistralai/devstral-2-123b-instruct-2512": { "id": "mistralai/devstral-2-123b-instruct-2512", "name": "Devstral 2 123B", "description": "Mistral coding agent model for repository tasks and software engineering workflows", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 1.4 } }, "mistralai/codestral-2508": { "id": "mistralai/codestral-2508", "name": "Codestral 2508", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "codestral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-01", "last_updated": "2025-08-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.8999999999999999 } }, "mistralai/Mistral-Nemo-Instruct-2407": { "id": "mistralai/Mistral-Nemo-Instruct-2407", "name": "Mistral Nemo", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.1003, "output": 0.1207 } }, "bytedance-seed/seed-2.0-lite": { "id": "bytedance-seed/seed-2.0-lite", "name": "ByteDance Seed 2.0 Lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "release_date": "2026-03-10", "last_updated": "2026-03-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 131072 }, "cost": { "input": 0.25, "output": 2 } }, "anthracite-org/magnum-v2-72b": { "id": "anthracite-org/magnum-v2-72b", "name": "Magnum V2 72B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 2.006, "output": 2.992 } }, "anthracite-org/magnum-v4-72b": { "id": "anthracite-org/magnum-v4-72b", "name": "Magnum v4 72B", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 2.006, "output": 2.992 } }, "inflatebot/MN-12B-Mag-Mell-R1": { "id": "inflatebot/MN-12B-Mag-Mell-R1", "name": "Mag Mell R1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": { "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "name": "Nvidia Nemotron 3 Nano Omni", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 65536 }, "cost": { "input": 0.105, "output": 0.42 } }, "nvidia/Llama-3.1-Nemotron-70B-Instruct-HF": { "id": "nvidia/Llama-3.1-Nemotron-70B-Instruct-HF", "name": "Nvidia Nemotron 70b", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.357, "output": 0.408 } }, "nvidia/nvidia-nemotron-nano-9b-v2": { "id": "nvidia/nvidia-nemotron-nano-9b-v2", "name": "Nvidia Nemotron Nano 9B v2", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-18", "last_updated": "2025-08-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.17, "output": 0.68 } }, "nvidia/Llama-3_3-Nemotron-Super-49B-v1_5": { "id": "nvidia/Llama-3_3-Nemotron-Super-49B-v1_5", "name": "Nvidia Nemotron Super 49B v1.5", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.05, "output": 0.25 } }, "nvidia/nemotron-3-nano-30b-a3b": { "id": "nvidia/nemotron-3-nano-30b-a3b", "name": "Nvidia Nemotron 3 Nano 30B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 262144 }, "cost": { "input": 0.17, "output": 0.68 } }, "nvidia/Llama-3.3-Nemotron-Super-49B-v1": { "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1", "name": "Nvidia Nemotron Super 49B", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.15, "output": 0.15 } }, "nvidia/nemotron-3-super-120b-a12b:thinking": { "id": "nvidia/nemotron-3-super-120b-a12b:thinking", "name": "Nvidia Nemotron 3 Super 120B Thinking", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.05, "output": 0.25 } }, "nvidia/nemotron-3-super-120b-a12b": { "id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nvidia Nemotron 3 Super 120B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.05, "output": 0.25 } }, "cognitivecomputations/dolphin-2.9.2-qwen2-72b": { "id": "cognitivecomputations/dolphin-2.9.2-qwen2-72b", "name": "Dolphin 72b", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 4096 }, "cost": { "input": 0.306, "output": 0.306 } }, "xiaomi/mimo-v2-flash-original": { "id": "xiaomi/mimo-v2-flash-original", "name": "MiMo V2 Flash Original", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.102, "output": 0.306 } }, "xiaomi/mimo-v2-flash-thinking-original": { "id": "xiaomi/mimo-v2-flash-thinking-original", "name": "MiMo V2 Flash (Thinking) Original", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.102, "output": 0.306 } }, "xiaomi/mimo-v2.5": { "id": "xiaomi/mimo-v2.5", "name": "MiMo V2.5", "description": "MiMo omni model for text, image, video, audio, and agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 131072 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "xiaomi/mimo-v2-omni": { "id": "xiaomi/mimo-v2-omni", "name": "MiMo V2 Omni", "description": "MiMo omni model for text, image, video, audio, and agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text", "image", "video", "audio"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 65536 }, "cost": { "input": 0.4, "output": 2, "cache_read": 0.08 } }, "xiaomi/mimo-v2-flash": { "id": "xiaomi/mimo-v2-flash", "name": "MiMo V2 Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.102, "output": 0.306 } }, "xiaomi/mimo-v2-pro": { "id": "xiaomi/mimo-v2-pro", "name": "MiMo V2 Pro", "description": "MiMo pro model for strong multimodal reasoning and agent execution", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-19", "last_updated": "2026-03-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 131072 }, "cost": { "input": 1, "output": 3, "cache_read": 0.2 } }, "xiaomi/mimo-v2.5-pro": { "id": "xiaomi/mimo-v2.5-pro", "name": "MiMo V2.5 Pro", "description": "MiMo pro model for strong multimodal reasoning and agent execution", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 131072 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.0036 } }, "xiaomi/mimo-v2-flash-thinking": { "id": "xiaomi/mimo-v2-flash-thinking", "name": "MiMo V2 Flash (Thinking)", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.102, "output": 0.306 } }, "anthropic/claude-haiku-latest": { "id": "anthropic/claude-haiku-latest", "name": "Claude Haiku Latest", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-29", "last_updated": "2026-03-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1 } }, "anthropic/claude-opus-4.7:thinking": { "id": "anthropic/claude-opus-4.7:thinking", "name": "Claude 4.7 Opus Thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007, "cache_read": 0.4998 } }, "anthropic/claude-opus-4.6:thinking:max": { "id": "anthropic/claude-opus-4.6:thinking:max", "name": "Claude 4.6 Opus Thinking Max", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007 } }, "anthropic/claude-opus-4.6:thinking:low": { "id": "anthropic/claude-opus-4.6:thinking:low", "name": "Claude 4.6 Opus Thinking Low", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007 } }, "anthropic/claude-sonnet-4.6:thinking": { "id": "anthropic/claude-sonnet-4.6:thinking", "name": "Claude Sonnet 4.6 Thinking", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 2.992, "output": 14.993999999999998 } }, "anthropic/claude-opus-4.7": { "id": "anthropic/claude-opus-4.7", "name": "Claude 4.7 Opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007, "cache_read": 0.4998 } }, "anthropic/claude-opus-4.6:thinking:medium": { "id": "anthropic/claude-opus-4.6:thinking:medium", "name": "Claude 4.6 Opus Thinking Medium", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007 } }, "anthropic/claude-opus-4.8:thinking": { "id": "anthropic/claude-opus-4.8:thinking", "name": "Claude Opus 4.8 Thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007, "cache_read": 0.4998 } }, "anthropic/claude-sonnet-latest": { "id": "anthropic/claude-sonnet-latest", "name": "Claude Sonnet Latest", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 2.992, "output": 14.994, "cache_read": 0.2992 } }, "anthropic/claude-opus-4.8": { "id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007, "cache_read": 0.4998 } }, "anthropic/claude-opus-latest": { "id": "anthropic/claude-opus-latest", "name": "Claude Opus Latest", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-29", "last_updated": "2026-03-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007, "cache_read": 0.4998 } }, "anthropic/claude-sonnet-4.6": { "id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-02-17", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 2.992, "output": 14.993999999999998 } }, "anthropic/claude-opus-4.6:thinking": { "id": "anthropic/claude-opus-4.6:thinking", "name": "Claude 4.6 Opus Thinking", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007 } }, "anthropic/claude-opus-4.6": { "id": "anthropic/claude-opus-4.6", "name": "Claude 4.6 Opus", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 128000 }, "cost": { "input": 4.998, "output": 25.007 } }, "tencent/Hunyuan-MT-7B": { "id": "tencent/Hunyuan-MT-7B", "name": "Hunyuan MT 7B", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "hunyuan", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-18", "last_updated": "2025-09-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 8192 }, "cost": { "input": 10, "output": 20 } }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Tencent: Hy3 preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 262144 }, "cost": { "input": 0.066, "output": 0.26, "cache_read": 0.029 } }, "chutesai/Mistral-Small-3.2-24B-Instruct-2506": { "id": "chutesai/Mistral-Small-3.2-24B-Instruct-2506", "name": "Mistral Small 3.2 24b Instruct", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "chutesai", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 131072 }, "cost": { "input": 0.2, "output": 0.4 } }, "dmind/dmind-1-mini": { "id": "dmind/dmind-1-mini", "name": "DMind-1-Mini", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.2, "output": 0.4 } }, "dmind/dmind-1": { "id": "dmind/dmind-1", "name": "DMind-1", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-01", "last_updated": "2025-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.3, "output": 0.6 } }, "MarinaraSpaghetti/NemoMix-Unleashed-12B": { "id": "MarinaraSpaghetti/NemoMix-Unleashed-12B", "name": "NemoMix 12B Unleashed", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "deepcogito/cogito-v1-preview-qwen-32B": { "id": "deepcogito/cogito-v1-preview-qwen-32B", "name": "Cogito v1 Preview Qwen 32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-10", "last_updated": "2025-05-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 32768 }, "cost": { "input": 1.7999999999999998, "output": 1.7999999999999998 } }, "cohere/command-r": { "id": "cohere/command-r", "name": "Cohere: Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-03-11", "last_updated": "2024-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 0.476, "output": 1.428 } }, "cohere/command-r-plus-08-2024": { "id": "cohere/command-r-plus-08-2024", "name": "Cohere: Command R+", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 4096 }, "cost": { "input": 2.856, "output": 14.246 } }, "stepfun/step-3.7-flash:thinking": { "id": "stepfun/step-3.7-flash:thinking", "name": "Step 3.7 Flash Thinking", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 256000 }, "cost": { "input": 0.2, "output": 1.15, "cache_read": 0.04 } }, "nex-agi/deepseek-v3.1-nex-n1": { "id": "nex-agi/deepseek-v3.1-nex-n1", "name": "DeepSeek V3.1 Nex N1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-10", "last_updated": "2025-12-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.27999999999999997, "output": 0.42000000000000004 } }, "undi95/remm-slerp-l2-13b": { "id": "undi95/remm-slerp-l2-13b", "name": "ReMM SLERP 13B", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 6144, "input": 6144, "output": 4096 }, "cost": { "input": 0.7989999999999999, "output": 1.2069999999999999 } }, "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16": { "id": "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16", "name": "Llama 3.1 70B Celeste v0.1", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "zai-org/glm-4.7-flash-original:thinking": { "id": "zai-org/glm-4.7-flash-original:thinking", "name": "GLM 4.7 Flash Original Thinking", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 0.07, "output": 0.4 } }, "zai-org/glm-4.7": { "id": "zai-org/glm-4.7", "name": "GLM 4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-01-29", "last_updated": "2026-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.8 } }, "zai-org/GLM-4.5-Air:thinking": { "id": "zai-org/GLM-4.5-Air:thinking", "name": "GLM 4.5 Air (Thinking)", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 98304 }, "cost": { "input": 0.12, "output": 0.8 } }, "zai-org/GLM-4.5:thinking": { "id": "zai-org/GLM-4.5:thinking", "name": "GLM 4.5 (Thinking)", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.3 } }, "zai-org/glm-4.5": { "id": "zai-org/glm-4.5", "name": "GLM 4.5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.3 } }, "zai-org/glm-5-original": { "id": "zai-org/glm-5-original", "name": "GLM 5 Original", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "zai-org/glm-5.1": { "id": "zai-org/glm-5.1", "name": "GLM 5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "input": 200000, "output": 131072 }, "cost": { "input": 0.3, "output": 2.55 } }, "zai-org/glm-4.7-original": { "id": "zai-org/glm-4.7-original", "name": "GLM 4.7 Original", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 65535 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "zai-org/glm-4.7-original:thinking": { "id": "zai-org/glm-4.7-original:thinking", "name": "GLM 4.7 Original Thinking", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 65535 }, "cost": { "input": 0.6, "output": 2.2, "cache_read": 0.11 } }, "zai-org/glm-4.7-flash-original": { "id": "zai-org/glm-4.7-flash-original", "name": "GLM 4.7 Flash Original", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 0.07, "output": 0.4 } }, "zai-org/glm-4.7:thinking": { "id": "zai-org/glm-4.7:thinking", "name": "GLM 4.7 Thinking", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 65535 }, "cost": { "input": 0.2, "output": 0.8 } }, "zai-org/glm-5.1:thinking": { "id": "zai-org/glm-5.1:thinking", "name": "GLM 5.1 Thinking", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "input": 200000, "output": 131072 }, "cost": { "input": 0.3, "output": 2.55 } }, "zai-org/glm-5:thinking": { "id": "zai-org/glm-5:thinking", "name": "GLM 5 Thinking", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 0.3, "output": 2.55 } }, "zai-org/glm-4.6v": { "id": "zai-org/glm-4.6v", "name": "GLM 4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 24000 }, "cost": { "input": 0.3, "output": 0.9 } }, "zai-org/glm-4.6-original": { "id": "zai-org/glm-4.6-original", "name": "GLM 4.6 Original", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": false, "structured_output": false, "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 65535 }, "cost": { "input": 0.35, "output": 1.4 } }, "zai-org/glm-5-original:thinking": { "id": "zai-org/glm-5-original:thinking", "name": "GLM 5 Original Thinking", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.2 } }, "zai-org/glm-4.7-flash:thinking": { "id": "zai-org/glm-4.7-flash:thinking", "name": "GLM 4.7 Flash Thinking", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 0.07, "output": 0.4 } }, "zai-org/glm-4.6v-flash-original": { "id": "zai-org/glm-4.6v-flash-original", "name": "GLM 4.6V Flash", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 24000 }, "cost": { "input": 0.1, "output": 0.4 } }, "zai-org/glm-4.6v-original": { "id": "zai-org/glm-4.6v-original", "name": "GLM 4.6V Original", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 24000 }, "cost": { "input": 0.6, "output": 0.9 } }, "zai-org/glm-latest": { "id": "zai-org/glm-latest", "name": "GLM Latest", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-03", "last_updated": "2026-05-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 131072 }, "cost": { "input": 0.75, "output": 2.6, "cache_read": 0.15 } }, "zai-org/glm-4.7-flash": { "id": "zai-org/glm-4.7-flash", "name": "GLM 4.7 Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 0.07, "output": 0.4 } }, "zai-org/GLM-4.6-turbo:thinking": { "id": "zai-org/GLM-4.6-turbo:thinking", "name": "GLM 4.6 Turbo (Thinking)", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-10-02", "last_updated": "2025-10-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 204800 }, "cost": { "input": 1, "output": 3 } }, "zai-org/GLM-4.5-Air": { "id": "zai-org/GLM-4.5-Air", "name": "GLM 4.5 Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 98304 }, "cost": { "input": 0.12, "output": 0.8 } }, "zai-org/GLM-4.6-turbo": { "id": "zai-org/GLM-4.6-turbo", "name": "GLM 4.6 Turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-10-02", "last_updated": "2025-10-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 204800 }, "cost": { "input": 1, "output": 3 } }, "zai-org/glm-5": { "id": "zai-org/glm-5", "name": "GLM 5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 200000, "input": 200000, "output": 128000 }, "cost": { "input": 0.3, "output": 2.55 } }, "Infermatic/MN-12B-Inferor-v0.0": { "id": "Infermatic/MN-12B-Inferor-v0.0", "name": "Mistral Nemo Inferor 12B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.25499999999999995, "output": 0.49299999999999994 } }, "shisa-ai/shisa-v2.1-llama3.3-70b": { "id": "shisa-ai/shisa-v2.1-llama3.3-70b", "name": "Shisa V2.1 Llama 3.3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 4096 }, "cost": { "input": 0.5, "output": 0.5 } }, "shisa-ai/shisa-v2-llama3.3-70b": { "id": "shisa-ai/shisa-v2-llama3.3-70b", "name": "Shisa V2 Llama 3.3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 0.5, "output": 0.5 } }, "abacusai/Dracarys-72B-Instruct": { "id": "abacusai/Dracarys-72B-Instruct", "name": "Llama 3.1 70B Dracarys 2", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-02", "last_updated": "2025-08-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "deepseek-ai/DeepSeek-V3.1:thinking": { "id": "deepseek-ai/DeepSeek-V3.1:thinking", "name": "DeepSeek V3.1 Thinking", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek-thinking", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.2, "output": 0.7 } }, "deepseek-ai/DeepSeek-V3.1-Terminus": { "id": "deepseek-ai/DeepSeek-V3.1-Terminus", "name": "DeepSeek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-08-02", "last_updated": "2025-08-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.25, "output": 0.7 } }, "deepseek-ai/deepseek-v3.2-exp-thinking": { "id": "deepseek-ai/deepseek-v3.2-exp-thinking", "name": "DeepSeek V3.2 Exp Thinking", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "input": 163840, "output": 65536 }, "cost": { "input": 0.27999999999999997, "output": 0.42000000000000004 } }, "deepseek-ai/DeepSeek-R1-0528": { "id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 163840 }, "cost": { "input": 0.4, "output": 1.7 } }, "deepseek-ai/DeepSeek-V3.1": { "id": "deepseek-ai/DeepSeek-V3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.2, "output": 0.7 } }, "deepseek-ai/deepseek-v3.2-exp": { "id": "deepseek-ai/deepseek-v3.2-exp", "name": "DeepSeek V3.2 Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163840, "input": 163840, "output": 65536 }, "cost": { "input": 0.27999999999999997, "output": 0.42000000000000004 } }, "deepseek-ai/DeepSeek-V3.1-Terminus:thinking": { "id": "deepseek-ai/DeepSeek-V3.1-Terminus:thinking", "name": "DeepSeek V3.1 Terminus (Thinking)", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek-thinking", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.25, "output": 0.7 } }, "arcee-ai/trinity-large-thinking": { "id": "arcee-ai/trinity-large-thinking", "name": "Trinity Large Thinking", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 80000 }, "cost": { "input": 0.25, "output": 0.9 } }, "arcee-ai/trinity-mini": { "id": "arcee-ai/trinity-mini", "name": "Trinity Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "trinity-mini", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 8192 }, "cost": { "input": 0.045000000000000005, "output": 0.15 } }, "soob3123/GrayLine-Qwen3-8B": { "id": "soob3123/GrayLine-Qwen3-8B", "name": "Grayline Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 32768 }, "cost": { "input": 0.3, "output": 0.3 } }, "soob3123/Veiled-Calla-12B": { "id": "soob3123/Veiled-Calla-12B", "name": "Veiled Calla 12B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-13", "last_updated": "2025-04-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.3, "output": 0.3 } }, "soob3123/amoral-gemma3-27B-v2": { "id": "soob3123/amoral-gemma3-27B-v2", "name": "Amoral Gemma3 27B v2", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-05-23", "last_updated": "2025-05-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 8192 }, "cost": { "input": 0.3, "output": 0.3 } }, "meganova-ai/manta-mini-1.0": { "id": "meganova-ai/manta-mini-1.0", "name": "Manta Mini 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-20", "last_updated": "2025-12-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 8192 }, "cost": { "input": 0.02, "output": 0.16 } }, "meganova-ai/manta-flash-1.0": { "id": "meganova-ai/manta-flash-1.0", "name": "Manta Flash 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-20", "last_updated": "2025-12-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.02, "output": 0.16 } }, "meganova-ai/manta-pro-1.0": { "id": "meganova-ai/manta-pro-1.0", "name": "Manta Pro 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-20", "last_updated": "2025-12-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.060000000000000005, "output": 0.5 } }, "qwen/Qwen3.6-35B-A3B:thinking": { "id": "qwen/Qwen3.6-35B-A3B:thinking", "name": "Qwen3.6 35B A3B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-04-19", "last_updated": "2026-04-19", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.112, "output": 0.8 } }, "qwen/qwen3-coder-plus": { "id": "qwen/qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-17", "last_updated": "2025-09-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 1, "output": 5 } }, "qwen/Qwen3.6-35B-A3B": { "id": "qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 16384 }, "cost": { "input": 0.112, "output": 0.8 } }, "qwen/qwen3-32b": { "id": "qwen/qwen3-32b", "name": "Qwen 3 32b", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 41000, "input": 41000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.3 } }, "qwen/Qwen3-VL-235B-A22B-Instruct": { "id": "qwen/Qwen3-VL-235B-A22B-Instruct", "name": "Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 262144 }, "cost": { "input": 0.3, "output": 1.2 } }, "qwen/qwen3-max": { "id": "qwen/qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 1.08018, "output": 5.4009 } }, "qwen/qwen3.5-397b-a17b-thinking": { "id": "qwen/qwen3.5-397b-a17b-thinking", "name": "Qwen3.5 397B A17B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 258048, "input": 258048, "output": 65536 }, "cost": { "input": 0.6, "output": 3.6 } }, "qwen/Qwen3-Next-80B-A3B-Instruct": { "id": "qwen/Qwen3-Next-80B-A3B-Instruct", "name": "Qwen3 Next 80B A3B (Instruct)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-09-11", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 262144 }, "cost": { "input": 0.15, "output": 0.65 } }, "qwen/Qwen3-235B-A22B-Instruct-2507-TEE": { "id": "qwen/Qwen3-235B-A22B-Instruct-2507-TEE", "name": "Qwen 3 235b A22B 2507 (TEE)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 262144 }, "cost": { "input": 0.13, "output": 0.5 } }, "qwen/qwq-32b-preview": { "id": "qwen/qwq-32b-preview", "name": "Qwen QwQ 32B Preview", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 32768 }, "cost": { "input": 0.2, "output": 0.2 } }, "qwen/qwen3-next-80b-a3b-thinking": { "id": "qwen/qwen3-next-80b-a3b-thinking", "name": "Qwen3 Next 80B A3B (Thinking)", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.65 } }, "qwen/qwen3-coder": { "id": "qwen/qwen3-coder", "name": "Qwen 3 Coder 480B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "input": 262000, "output": 65536 }, "cost": { "input": 0.13, "output": 0.5 } }, "qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen 3 235b A22B 2507 Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-11", "last_updated": "2025-09-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 262144 }, "cost": { "input": 0.3, "output": 0.5 } }, "qwen/qwen3.5-plus": { "id": "qwen/qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983616, "input": 983616, "output": 65536 }, "cost": { "input": 0.4, "output": 2.4, "cache_read": 0.04 } }, "qwen/qwen3.5-plus-thinking": { "id": "qwen/qwen3.5-plus-thinking", "name": "Qwen3.5 Plus Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 983616, "input": 983616, "output": 65536 }, "cost": { "input": 0.4, "output": 2.4, "cache_read": 0.04 } }, "qwen/qwen3-235b-a22b": { "id": "qwen/qwen3-235b-a22b", "name": "Qwen 3 235b A22B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-04-29", "last_updated": "2025-04-29", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 41000, "input": 41000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.5 } }, "qwen/qwen-2.5-72b-instruct": { "id": "qwen/qwen-2.5-72b-instruct", "name": "Qwen2.5 72B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-03", "last_updated": "2025-07-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 8192 }, "cost": { "input": 0.357, "output": 0.408 } }, "qwen/qwen3-coder-next": { "id": "qwen/qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 65536 }, "cost": { "input": 0.15, "output": 1.5 } }, "qwen/qwen3.5-9b": { "id": "qwen/qwen3.5-9b", "name": "Qwen3.5 9B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 81920 } ], "tool_call": false, "structured_output": false, "release_date": "2026-03-10", "last_updated": "2026-03-10", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 65536 }, "cost": { "input": 0.05, "output": 0.15 } }, "qwen/qwen3.5-397b-a17b": { "id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5 397B A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 258048, "input": 258048, "output": 65536 }, "cost": { "input": 0.6, "output": 3.6 } }, "qwen/qwen3-coder-flash": { "id": "qwen/qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-17", "last_updated": "2025-09-17", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65536 }, "cost": { "input": 0.3, "output": 1.5 } }, "qwen/Qwen2.5-Coder-32B-Instruct": { "id": "qwen/Qwen2.5-Coder-32B-Instruct", "name": "Qwen 2.5 Coder 32b", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-03", "last_updated": "2025-07-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32000, "input": 32000, "output": 8192 }, "cost": { "input": 0.2006, "output": 0.2006 } }, "qwen/Qwen3-8B": { "id": "qwen/Qwen3-8B", "name": "Qwen 3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 41000, "input": 41000, "output": 32768 }, "cost": { "input": 0.47, "output": 0.47 } }, "qwen/qwen3-14b": { "id": "qwen/qwen3-14b", "name": "Qwen 3 14b", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-01-01", "last_updated": "2024-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 41000, "input": 41000, "output": 32768 }, "cost": { "input": 0.08, "output": 0.24 } }, "qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen 3 235b A22B 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 262144 }, "cost": { "input": 0.13, "output": 0.5 } }, "qwen/qwen3-30b-a3b": { "id": "qwen/qwen3-30b-a3b", "name": "Qwen3 30B A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-27", "last_updated": "2025-02-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 41000, "input": 41000, "output": 32768 }, "cost": { "input": 0.1, "output": 0.3 } }, "amazon/nova-lite-v1": { "id": "amazon/nova-lite-v1", "name": "Amazon Nova Lite 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-lite", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "input": 300000, "output": 5120 }, "cost": { "input": 0.0595, "output": 0.238 } }, "amazon/nova-pro-v1": { "id": "amazon/nova-pro-v1", "name": "Amazon Nova Pro 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova-pro", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 300000, "input": 300000, "output": 32000 }, "cost": { "input": 0.7989999999999999, "output": 3.1959999999999997 } }, "amazon/nova-micro-v1": { "id": "amazon/nova-micro-v1", "name": "Amazon Nova Micro 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-micro", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 5120 }, "cost": { "input": 0.0357, "output": 0.1394 } }, "amazon/nova-2-lite-v1": { "id": "amazon/nova-2-lite-v1", "name": "Amazon Nova 2 Lite", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-03", "last_updated": "2024-12-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 65535 }, "cost": { "input": 0.5099999999999999, "output": 4.25 } }, "alibaba/qwen3.6-flash": { "id": "alibaba/qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 991800, "output": 65536 }, "cost": { "input": 0.19, "output": 1.16 } }, "alibaba/qwen3.6-27b": { "id": "alibaba/qwen3.6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.203, "output": 2.24 } }, "alibaba/qwen3.6-27b:thinking": { "id": "alibaba/qwen3.6-27b:thinking", "name": "Qwen3.6 27B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 131072 } ], "tool_call": false, "structured_output": false, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 260096, "input": 260096, "output": 65536 }, "cost": { "input": 0.203, "output": 2.24 } }, "aion-labs/aion-rp-llama-3.1-8b": { "id": "aion-labs/aion-rp-llama-3.1-8b", "name": "Llama 3.1 8b (uncensored)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 16384 }, "cost": { "input": 0.2006, "output": 0.2006 } }, "aion-labs/aion-2.5": { "id": "aion-labs/aion-2.5", "name": "AionLabs: Aion-2.5", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 32768 }, "cost": { "input": 1, "output": 3, "cache_read": 0.35 } }, "aion-labs/aion-1.0-mini": { "id": "aion-labs/aion-1.0-mini", "name": "Aion 1.0 mini (DeepSeek)", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-20", "last_updated": "2025-02-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 8192 }, "cost": { "input": 0.7989999999999999, "output": 1.394 } }, "aion-labs/aion-2.0": { "id": "aion-labs/aion-2.0", "name": "AionLabs: Aion-2.0", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 32768 }, "cost": { "input": 0.8, "output": 1.6 } }, "aion-labs/aion-1.0": { "id": "aion-labs/aion-1.0", "name": "Aion 1.0", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-01", "last_updated": "2025-02-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 8192 }, "cost": { "input": 3.995, "output": 7.99 } }, "pamanseau/OpenReasoning-Nemotron-32B": { "id": "pamanseau/OpenReasoning-Nemotron-32B", "name": "OpenReasoning Nemotron 32B", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "input": 32768, "output": 65536 }, "cost": { "input": 0.1, "output": 0.4 } }, "LatitudeGames/Wayfarer-Large-70B-Llama-3.3": { "id": "LatitudeGames/Wayfarer-Large-70B-Llama-3.3", "name": "Llama 3.3 70B Wayfarer", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-20", "last_updated": "2025-02-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.700000007, "output": 0.700000007 } }, "baseten/Kimi-K2-Instruct-FP4": { "id": "baseten/Kimi-K2-Instruct-FP4", "name": "Kimi K2 0711 Instruct FP4", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-11", "last_updated": "2025-07-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 131072 }, "cost": { "input": 0.1, "output": 2 } }, "inflection/inflection-3-pi": { "id": "inflection/inflection-3-pi", "name": "Inflection 3 Pi", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-10-11", "last_updated": "2024-10-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "input": 8000, "output": 4096 }, "cost": { "input": 2.499, "output": 9.996 } }, "inflection/inflection-3-productivity": { "id": "inflection/inflection-3-productivity", "name": "Inflection 3 Productivity", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-10-11", "last_updated": "2024-10-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8000, "input": 8000, "output": 4096 }, "cost": { "input": 2.499, "output": 9.996 } }, "ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0": { "id": "ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0", "name": "Omega Directive 24B Unslop v2.0", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 32768 }, "cost": { "input": 0.5, "output": 0.5 } }, "MiniMaxAI/MiniMax-M1-80k": { "id": "MiniMaxAI/MiniMax-M1-80k", "name": "MiniMax M1 80K", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-06-16", "last_updated": "2025-06-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 1000000, "output": 131072 }, "cost": { "input": 0.6052, "output": 2.4225000000000003 } }, "upstage/solar-pro-3": { "id": "upstage/solar-pro-3", "name": "Solar Pro 3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 128000 }, "cost": { "input": 0.15, "output": 0.6, "cache_read": 0.015 } }, "allenai/olmo-3-32b-think": { "id": "allenai/olmo-3-32b-think", "name": "Olmo 3 32B Think", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "allenai", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-11-01", "last_updated": "2025-11-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.3, "output": 0.44999999999999996 } }, "essentialai/rnj-1-instruct": { "id": "essentialai/rnj-1-instruct", "name": "RNJ-1 Instruct 8B", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "rnj", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-13", "last_updated": "2025-12-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 8192 }, "cost": { "input": 0.15, "output": 0.15 } }, "deepseek/deepseek-v4-flash": { "id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "deepseek/deepseek-v4-flash:thinking": { "id": "deepseek/deepseek-v4-flash:thinking", "name": "DeepSeek V4 Flash (Thinking)", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.028 } }, "deepseek/deepseek-v4-pro-cheaper:thinking": { "id": "deepseek/deepseek-v4-pro-cheaper:thinking", "name": "DeepSeek V4 Pro Cheaper (Thinking)", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-25", "last_updated": "2026-04-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "deepseek/deepseek-prover-v2-671b": { "id": "deepseek/deepseek-prover-v2-671b", "name": "DeepSeek Prover v2 671B", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-04-30", "last_updated": "2025-04-30", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 160000, "input": 160000, "output": 16384 }, "cost": { "input": 1, "output": 2.5 } }, "deepseek/deepseek-latest": { "id": "deepseek/deepseek-latest", "name": "DeepSeek Latest", "description": "DeepSeek chat model for instruction following, coding, and analysis", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-03", "last_updated": "2026-05-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 384000 }, "cost": { "input": 1.1, "output": 2.2, "cache_read": 0.11 } }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 384000 }, "cost": { "input": 1.1, "output": 2.2, "cache_read": 0.11 } }, "deepseek/deepseek-v4-pro:thinking": { "id": "deepseek/deepseek-v4-pro:thinking", "name": "DeepSeek V4 Pro (Thinking)", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 384000 }, "cost": { "input": 1.1, "output": 2.2, "cache_read": 0.11 } }, "deepseek/deepseek-v3.2-speciale": { "id": "deepseek/deepseek-v3.2-speciale", "name": "DeepSeek V3.2 Speciale", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163000, "input": 163000, "output": 65536 }, "cost": { "input": 0.27999999999999997, "output": 0.42000000000000004 } }, "deepseek/deepseek-v4-pro-cheaper": { "id": "deepseek/deepseek-v4-pro-cheaper", "name": "DeepSeek V4 Pro Cheaper", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "xhigh"] } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-25", "last_updated": "2026-04-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "input": 1048576, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "deepseek/deepseek-v3.2:thinking": { "id": "deepseek/deepseek-v3.2:thinking", "name": "DeepSeek V3.2 Thinking", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163000, "input": 163000, "output": 65536 }, "cost": { "input": 0.27999999999999997, "output": 0.42000000000000004 } }, "deepseek/deepseek-v3.2": { "id": "deepseek/deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 163000, "input": 163000, "output": 65536 }, "cost": { "input": 0.27999999999999997, "output": 0.42000000000000004 } }, "minimax/minimax-m3:thinking": { "id": "minimax/minimax-m3:thinking", "name": "MiniMax M3 Thinking", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 512000, "input": 512000, "output": 80000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m2.5": { "id": "minimax/minimax-m2.5", "name": "MiniMax M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "input": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "minimax/minimax-m2.1": { "id": "minimax/minimax-m2.1", "name": "MiniMax M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2025-12-19", "last_updated": "2025-12-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 200000, "output": 131072 }, "cost": { "input": 0.33, "output": 1.32 } }, "minimax/minimax-m2.7-turbo": { "id": "minimax/minimax-m2.7-turbo", "name": "MiniMax M2.7 Turbo", "description": "Efficient MiniMax model for quick assistance, coding, and routine automation", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "input": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4 } }, "minimax/minimax-01": { "id": "minimax/minimax-01", "name": "MiniMax 01", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-01-15", "last_updated": "2025-01-15", "modalities": { "input": ["text", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000192, "input": 1000192, "output": 16384 }, "cost": { "input": 0.1394, "output": 1.1219999999999999 } }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", "name": "MiniMax M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 512000, "input": 512000, "output": 80000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m2-her": { "id": "minimax/minimax-m2-her", "name": "MiniMax M2-her", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-01-24", "last_updated": "2026-01-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65532, "input": 65532, "output": 2048 }, "cost": { "input": 0.30200000000000005, "output": 1.2069999999999999 } }, "minimax/minimax-latest": { "id": "minimax/minimax-latest", "name": "MiniMax Latest", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-03", "last_updated": "2026-05-03", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 512000, "input": 512000, "output": 80000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06 } }, "minimax/minimax-m2.7": { "id": "minimax/minimax-m2.7", "name": "MiniMax M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 204800, "input": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "kwaipilot/kat-coder-pro-v2": { "id": "kwaipilot/kat-coder-pro-v2", "name": "KAT Coder Pro V2", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-03-28", "last_updated": "2026-03-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 256000, "output": 80000 }, "cost": { "input": 0.3, "output": 1.2 } }, "mlabonne/NeuralDaredevil-8B-abliterated": { "id": "mlabonne/NeuralDaredevil-8B-abliterated", "name": "Neural Daredevil 8B abliterated", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 8192, "input": 8192, "output": 8192 }, "cost": { "input": 0.44, "output": 0.44 } }, "VongolaChouko/Starcannon-Unleashed-12B-v1.0": { "id": "VongolaChouko/Starcannon-Unleashed-12B-v1.0", "name": "Mistral Nemo Starcannon 12b v1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "TEE/minimax-m2.5": { "id": "TEE/minimax-m2.5", "name": "MiniMax M2.5 TEE", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 196608, "input": 196608, "output": 131072 }, "cost": { "input": 0.2, "output": 1.38 } }, "TEE/glm-4.7": { "id": "TEE/glm-4.7", "name": "GLM 4.7 TEE", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-01-29", "last_updated": "2026-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131000, "input": 131000, "output": 65535 }, "cost": { "input": 0.85, "output": 3.3 } }, "TEE/llama3-3-70b": { "id": "TEE/llama3-3-70b", "name": "Llama 3.3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-03", "last_updated": "2025-07-03", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 2, "output": 2 } }, "TEE/gemma-4-31b-it": { "id": "TEE/gemma-4-31b-it", "name": "Gemma 4 31B IT TEE", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "release_date": "2026-05-26", "last_updated": "2026-05-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 262144 }, "cost": { "input": 0.15, "output": 0.46 } }, "TEE/qwen3-30b-a3b-instruct-2507": { "id": "TEE/qwen3-30b-a3b-instruct-2507", "name": "Qwen3 30B A3B Instruct 2507 TEE", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262000, "input": 262000, "output": 32768 }, "cost": { "input": 0.15, "output": 0.44999999999999996 } }, "TEE/glm-5.1": { "id": "TEE/glm-5.1", "name": "GLM 5.1 TEE", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "input": 202752, "output": 65535 }, "cost": { "input": 1.5, "output": 5.25, "cache_read": 0.3 } }, "TEE/deepseek-v4-pro": { "id": "TEE/deepseek-v4-pro", "name": "DeepSeek V4 Pro TEE", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "release_date": "2026-04-25", "last_updated": "2026-04-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 800000, "input": 800000, "output": 65536 }, "cost": { "input": 1.5, "output": 5.25, "cache_read": 0.15 } }, "TEE/deepseek-v4-pro:thinking": { "id": "TEE/deepseek-v4-pro:thinking", "name": "DeepSeek V4 Pro Thinking TEE", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 800000, "input": 800000, "output": 65536 }, "cost": { "input": 1.5, "output": 5.25, "cache_read": 0.15 } }, "TEE/gemma4-31b:thinking": { "id": "TEE/gemma4-31b:thinking", "name": "Gemma 4 31B Thinking TEE", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "release_date": "2026-05-02", "last_updated": "2026-05-02", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 131072 }, "cost": { "input": 0.45, "output": 1 } }, "TEE/kimi-k2.5": { "id": "TEE/kimi-k2.5", "name": "Kimi K2.5 TEE", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-01-29", "last_updated": "2026-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65535 }, "cost": { "input": 0.3, "output": 1.9 } }, "TEE/qwen2.5-vl-72b-instruct": { "id": "TEE/qwen2.5-vl-72b-instruct", "name": "Qwen2.5 VL 72B TEE", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-02-01", "last_updated": "2025-02-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 8192 }, "cost": { "input": 0.7, "output": 0.7 } }, "TEE/gpt-oss-120b": { "id": "TEE/gpt-oss-120b", "name": "GPT-OSS 120B TEE", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 16384 }, "cost": { "input": 2, "output": 2 } }, "TEE/qwen3.5-27b": { "id": "TEE/qwen3.5-27b", "name": "Qwen3.5 27B TEE", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-03-13", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2.4 } }, "TEE/gemma-4-26b-a4b-uncensored": { "id": "TEE/gemma-4-26b-a4b-uncensored", "name": "Gemma 4 26B A4B Uncensored TEE", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "release_date": "2026-05-23", "last_updated": "2026-05-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "input": 65536, "output": 65536 }, "cost": { "input": 0.15, "output": 0.7 } }, "TEE/deepseek-v3.1": { "id": "TEE/deepseek-v3.1", "name": "DeepSeek V3.1 TEE", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "input": 164000, "output": 8192 }, "cost": { "input": 1, "output": 2.5 } }, "TEE/kimi-k2.6": { "id": "TEE/kimi-k2.6", "name": "Kimi K2.6 TEE", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 65536 }, "cost": { "input": 1.5, "output": 5.25, "cache_read": 0.375 } }, "TEE/qwen3.5-397b-a17b": { "id": "TEE/qwen3.5-397b-a17b", "name": "Qwen3.5 397B A17B TEE", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-28", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 258048, "input": 258048, "output": 65536 }, "cost": { "input": 0.6, "output": 3.6 } }, "TEE/kimi-k2.5-thinking": { "id": "TEE/kimi-k2.5-thinking", "name": "Kimi K2.5 Thinking TEE", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "release_date": "2026-01-29", "last_updated": "2026-01-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 65535 }, "cost": { "input": 0.3, "output": 1.9 } }, "TEE/qwen3.6-35b-a3b-uncensored": { "id": "TEE/qwen3.6-35b-a3b-uncensored", "name": "Qwen3.6 35B A3B Uncensored TEE", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 131072 } ], "tool_call": true, "structured_output": true, "release_date": "2026-05-23", "last_updated": "2026-05-23", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 131072 }, "cost": { "input": 0.3, "output": 1.5 } }, "TEE/glm-4.7-flash": { "id": "TEE/glm-4.7-flash", "name": "GLM 4.7 Flash TEE", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 203000, "input": 203000, "output": 65535 }, "cost": { "input": 0.15, "output": 0.5 } }, "TEE/gemma4-31b": { "id": "TEE/gemma4-31b", "name": "Gemma 4 31B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "release_date": "2026-04-04", "last_updated": "2026-04-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 131072 }, "cost": { "input": 0.45, "output": 1 } }, "TEE/gemma-3-27b-it": { "id": "TEE/gemma-3-27b-it", "name": "Gemma 3 27B TEE", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-03-10", "last_updated": "2025-03-10", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 8192 }, "cost": { "input": 0.2, "output": 0.8 } }, "TEE/gpt-oss-20b": { "id": "TEE/gpt-oss-20b", "name": "GPT-OSS 20B TEE", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "input": 131072, "output": 8192 }, "cost": { "input": 0.2, "output": 0.8 } }, "TEE/glm-5": { "id": "TEE/glm-5", "name": "GLM 5 TEE", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 203000, "input": 203000, "output": 65535 }, "cost": { "input": 1.2, "output": 3.5 } }, "TEE/glm-5.1-thinking": { "id": "TEE/glm-5.1-thinking", "name": "GLM 5.1 Thinking TEE", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 202752, "input": 202752, "output": 65535 }, "cost": { "input": 1.5, "output": 5.25, "cache_read": 0.3 } }, "TEE/deepseek-v3.2": { "id": "TEE/deepseek-v3.2", "name": "DeepSeek V3.2 TEE", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 164000, "input": 164000, "output": 65536 }, "cost": { "input": 0.5, "output": 1 } }, "TEE/qwen3.5-122b-a10b": { "id": "TEE/qwen3.5-122b-a10b", "name": "Qwen3.5 122B A10B TEE", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": false, "release_date": "2026-05-26", "last_updated": "2026-05-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 262144, "input": 262144, "output": 262144 }, "cost": { "input": 0.46, "output": 3.68 } }, "Sao10K/L3.1-70B-Hanami-x1": { "id": "Sao10K/L3.1-70B-Hanami-x1", "name": "Llama 3.1 70B Hanami", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Sao10K/L3-8B-Stheno-v3.2": { "id": "Sao10K/L3-8B-Stheno-v3.2", "name": "Sao10K Stheno 8b", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-11-29", "last_updated": "2024-11-29", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 8192 }, "cost": { "input": 0.2006, "output": 0.2006 } }, "Sao10K/L3.3-70B-Euryale-v2.3": { "id": "Sao10K/L3.3-70B-Euryale-v2.3", "name": "Llama 3.3 70B Euryale", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 20480, "input": 20480, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Sao10K/L3.1-70B-Euryale-v2.2": { "id": "Sao10K/L3.1-70B-Euryale-v2.2", "name": "Llama 3.1 70B Euryale", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 20480, "input": 20480, "output": 16384 }, "cost": { "input": 0.306, "output": 0.357 } }, "Steelskull/L3.3-MS-Evalebis-70b": { "id": "Steelskull/L3.3-MS-Evalebis-70b", "name": "MS Evalebis 70b", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Steelskull/L3.3-MS-Nevoria-70b": { "id": "Steelskull/L3.3-MS-Nevoria-70b", "name": "Steelskull Nevoria 70b", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Steelskull/L3.3-Cu-Mai-R1-70b": { "id": "Steelskull/L3.3-Cu-Mai-R1-70b", "name": "Llama 3.3 70B Cu Mai", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Steelskull/L3.3-MS-Evayale-70B": { "id": "Steelskull/L3.3-MS-Evayale-70B", "name": "Evayale 70b ", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Steelskull/L3.3-Nevoria-R1-70b": { "id": "Steelskull/L3.3-Nevoria-R1-70b", "name": "Steelskull Nevoria R1 70b", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.49299999999999994, "output": 0.49299999999999994 } }, "Steelskull/L3.3-Electra-R1-70b": { "id": "Steelskull/L3.3-Electra-R1-70b", "name": "Steelskull Electra R1 70b", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 16384, "input": 16384, "output": 16384 }, "cost": { "input": 0.69989, "output": 0.69989 } } } }, "moark": { "id": "moark", "env": ["MOARK_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://moark.com/v1", "name": "Moark", "doc": "https://moark.com/docs/openapi/v1#tag/%E6%96%87%E6%9C%AC%E7%94%9F%E6%88%90", "models": { "MiniMax-M2.1": { "id": "MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 2.1, "output": 8.4 } }, "GLM-4.7": { "id": "GLM-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 3.5, "output": 14 } } } }, "lilac": { "id": "lilac", "env": ["LILAC_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.getlilac.com/v1", "name": "Lilac", "doc": "https://docs.getlilac.com/inference/models", "models": { "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.7, "output": 3.5, "cache_read": 0.2 } }, "minimaxai/minimax-m3": { "id": "minimaxai/minimax-m3", "name": "MiniMax M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax-m3", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 1048576 }, "cost": { "input": 0.28, "output": 1.1, "cache_read": 0.05 } }, "google/gemma-4-31b-it": { "id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262100, "output": 262100 }, "cost": { "input": 0.11, "output": 0.35 } }, "zai-org/glm-5.2": { "id": "zai-org/glm-5.2", "name": "GLM 5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 524288 }, "cost": { "input": 0.9, "output": 3, "cache_read": 0.27 } } } }, "ambient": { "id": "ambient", "env": ["AMBIENT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.ambient.xyz/v1", "name": "Ambient", "doc": "https://ambient.xyz", "models": { "moonshotai/kimi-k2.7-code": { "id": "moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.84, "output": 3.99, "cache_read": 0.16, "cache_write": 0 } }, "moonshotai/kimi-k2.6": { "id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.2, "cache_write": 0 } }, "zai-org/GLM-5.1-FP8": { "id": "zai-org/GLM-5.1-FP8", "name": "GLM 5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0, "cache_write": 0 } }, "zai-org/GLM-5.2-FP8": { "id": "zai-org/GLM-5.2-FP8", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 202752 }, "cost": { "input": 1.2, "output": 4.2, "cache_read": 0.26, "cache_write": 0 } } } }, "neon": { "id": "neon", "env": ["NEON_AI_GATEWAY_BASE_URL", "NEON_AI_GATEWAY_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/mlflow/v1", "name": "Neon", "doc": "https://neon.com/docs", "models": { "gemini-3-flash": { "id": "gemini-3-flash", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1 } }, "claude-sonnet-4": { "id": "claude-sonnet-4", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "qwen3-next-80b-a3b-instruct": { "id": "qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.15, "output": 1.2 } }, "llama-4-maverick": { "id": "llama-4-maverick", "name": "Llama 4 Maverick 17B Instruct", "description": "Open multimodal Llama for strong reasoning with efficient everyday serving", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 16384 }, "cost": { "input": 0.5, "output": 1.5 } }, "claude-opus-4-5": { "id": "claude-opus-4-5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5": { "id": "gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gemma-3-12b": { "id": "gemma-3-12b", "name": "Gemma 3 12B", "description": "Google's open-weight Gemma 3 vision-language model for text and image understanding", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.15, "output": 0.5 } }, "gpt-5-1-codex-mini": { "id": "gpt-5-1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gpt-5-1": { "id": "gpt-5-1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "meta-llama-3-3-70b-instruct": { "id": "meta-llama-3-3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 4096 }, "cost": { "input": 0.5, "output": 1.5 } }, "gemini-3-pro": { "id": "gemini-3-pro", "name": "Gemini 3 Pro Preview", "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-18", "last_updated": "2025-11-18", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "gpt-5-2": { "id": "gpt-5-2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gemini-3-1-flash-lite": { "id": "gemini-3-1-flash-lite", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5 } }, "claude-sonnet-4-5": { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-opus-4-7": { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gemini-2-5-pro": { "id": "gemini-2-5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [ { "input": 2.5, "output": 15, "cache_read": 0.25, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2.5, "output": 15, "cache_read": 0.25 } } }, "claude-opus-4-8": { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5-3-codex": { "id": "gpt-5-3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5-4-nano": { "id": "gpt-5-4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "claude-opus-4-1": { "id": "claude-opus-4-1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 31999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 32000 }, "cost": { "input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75 } }, "gpt-5-4-mini": { "id": "gpt-5-4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "gemini-2-5-flash": { "id": "gemini-2-5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 0, "max": 24576 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1 } }, "gpt-5-2-codex": { "id": "gpt-5-2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-oss-120b": { "id": "gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.072, "output": 0.28 } }, "gemini-3-5-flash": { "id": "gemini-3-5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "gemini-3-1-pro": { "id": "gemini-3-1-pro", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 65536 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "claude-haiku-4-5": { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "claude-opus-4-6": { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 127999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "qwen35-122b-a10b": { "id": "qwen35-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 8000 }, "cost": { "input": 0.22, "output": 2.2 } }, "meta-llama-3-1-8b-instruct": { "id": "meta-llama-3-1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Meta's compact open-weight Llama 3.1 model for fast, low-cost text generation", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.15, "output": 0.45 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gpt-5-nano": { "id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 0.05, "output": 0.4, "cache_read": 0.005 } }, "claude-sonnet-4-6": { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "budget_tokens", "min": 1024, "max": 63999 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 64000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gpt-5-4": { "id": "gpt-5-4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "gpt-5-1-codex-max": { "id": "gpt-5-1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": { "input": ["text", "image"], "output": ["text", "image"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "provider": { "npm": "@ai-sdk/openai", "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/openai/v1", "shape": "responses" }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "gpt-oss-20b": { "id": "gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.05, "output": 0.2 } } } }, "upstage": { "id": "upstage", "env": ["UPSTAGE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.upstage.ai/v1/solar", "name": "Upstage", "doc": "https://developers.upstage.ai/docs/apis/chat", "models": { "solar-pro2": { "id": "solar-pro2", "name": "solar-pro2", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "solar-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 65536, "output": 8192 }, "cost": { "input": 0.25, "output": 0.25 } }, "solar-pro3": { "id": "solar-pro3", "name": "solar-pro3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "solar-pro", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 131072, "output": 8192 }, "cost": { "input": 0.25, "output": 0.25 } }, "solar-mini": { "id": "solar-mini", "name": "solar-mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "solar-mini", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-06-12", "last_updated": "2025-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.15, "output": 0.15 } } } }, "zhipuai-coding-plan": { "id": "zhipuai-coding-plan", "env": ["ZHIPU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://open.bigmodel.cn/api/coding/paas/v4", "name": "Zhipu AI Coding Plan", "doc": "https://docs.bigmodel.cn/cn/coding-plan/overview", "models": { "glm-5.1": { "id": "glm-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": { "input": ["text", "image", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-5-turbo": { "id": "glm-5-turbo", "name": "GLM-5-Turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.5-air": { "id": "glm-4.5-air", "name": "GLM-4.5-Air", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 98304 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.6v": { "id": "glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32768 }, "cost": { "input": 0.3, "output": 0.9 } }, "glm-5.2": { "id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "glm-4.7": { "id": "glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "chutes": { "id": "chutes", "env": ["CHUTES_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://llm.chutes.ai/v1", "name": "Chutes", "doc": "https://llm.chutes.ai/v1/models", "models": { "moonshotai/Kimi-K2.6-TEE": { "id": "moonshotai/Kimi-K2.6-TEE", "name": "Kimi K2.6 TEE", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65535 }, "cost": { "input": 0.66, "output": 3.5, "cache_read": 0.33 } }, "moonshotai/Kimi-K2.5-TEE": { "id": "moonshotai/Kimi-K2.5-TEE", "name": "Kimi K2.5 TEE", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2024-10", "release_date": "2026-01", "last_updated": "2026-01", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65535 }, "cost": { "input": 0.44, "output": 2, "cache_read": 0.22 } }, "google/gemma-4-31B-turbo-TEE": { "id": "google/gemma-4-31B-turbo-TEE", "name": "gemma 4 31B turbo TEE", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 0.12, "output": 0.37, "cache_read": 0.06 } }, "Qwen/Qwen3-32B-TEE": { "id": "Qwen/Qwen3-32B-TEE", "name": "Qwen3 32B TEE", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 40960, "output": 40960 }, "cost": { "input": 0.104, "output": 0.416, "cache_read": 0.052 } }, "Qwen/Qwen3.6-27B-TEE": { "id": "Qwen/Qwen3.6-27B-TEE", "name": "Qwen3.6 27B TEE", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 2, "cache_read": 0.15 } }, "Qwen/Qwen3-235B-A22B-Thinking-2507-TEE": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507-TEE", "name": "Qwen3 235B A22B Thinking 2507 TEE", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07", "last_updated": "2026-06-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.2989, "output": 1.1957, "cache_read": 0.14945 } }, "Qwen/Qwen3.5-397B-A17B-TEE": { "id": "Qwen/Qwen3.5-397B-A17B-TEE", "name": "Qwen3.5 397B A17B TEE", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.45, "output": 3, "cache_read": 0.225 } }, "unsloth/Mistral-Nemo-Instruct-2407-TEE": { "id": "unsloth/Mistral-Nemo-Instruct-2407-TEE", "name": "Mistral Nemo Instruct 2407 TEE", "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.0245, "output": 0.0978, "cache_read": 0.01225 } }, "zai-org/GLM-5-TEE": { "id": "zai-org/GLM-5-TEE", "name": "GLM 5 TEE", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 65535 }, "cost": { "input": 0.95, "output": 2.55, "cache_read": 0.475 } }, "zai-org/GLM-5.1-TEE": { "id": "zai-org/GLM-5.1-TEE", "name": "GLM 5.1 TEE", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 65535 }, "cost": { "input": 0.98, "output": 3.08, "cache_read": 0.49 } }, "zai-org/GLM-5.2-TEE": { "id": "zai-org/GLM-5.2-TEE", "name": "GLM 5.2 TEE", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 65535 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 0.7 } }, "deepseek-ai/DeepSeek-V3.2-TEE": { "id": "deepseek-ai/DeepSeek-V3.2-TEE", "name": "DeepSeek V3.2 TEE", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2025-12", "last_updated": "2026-06-21", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 65536 }, "cost": { "input": 1, "output": 1, "cache_read": 0.5 } }, "MiniMaxAI/MiniMax-M2.5-TEE": { "id": "MiniMaxAI/MiniMax-M2.5-TEE", "name": "MiniMax M2.5 TEE", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 65536 }, "cost": { "input": 0.15, "output": 1.2, "cache_read": 0.075 } } } }, "minimax-cn-coding-plan": { "id": "minimax-cn-coding-plan", "env": ["MINIMAX_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://api.minimaxi.com/anthropic/v1", "name": "MiniMax Token Plan (minimaxi.com)", "doc": "https://platform.minimaxi.com/docs/token-plan/intro", "models": { "MiniMax-M2.1": { "id": "MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0 } }, "MiniMax-M2.5-highspeed": { "id": "MiniMax-M2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2.7-highspeed": { "id": "MiniMax-M2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2": { "id": "MiniMax-M2", "name": "MiniMax-M2", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M3": { "id": "MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } }, "MiniMax-M2.7": { "id": "MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0, "cache_write": 0 } } } }, "deepseek": { "id": "deepseek", "env": ["DEEPSEEK_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.deepseek.com", "name": "DeepSeek", "doc": "https://api-docs.deepseek.com/quick_start/pricing", "models": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.435, "output": 0.87, "cache_read": 0.003625 } }, "deepseek-reasoner": { "id": "deepseek-reasoner", "name": "DeepSeek Reasoner", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } }, "deepseek-chat": { "id": "deepseek-chat", "name": "DeepSeek Chat", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 384000 }, "cost": { "input": 0.14, "output": 0.28, "cache_read": 0.0028 } } } }, "wafer.ai": { "id": "wafer.ai", "env": ["WAFER_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://pass.wafer.ai/v1", "name": "Wafer", "doc": "https://docs.wafer.ai/wafer-pass", "models": { "glm5.2-fast": { "id": "glm5.2-fast", "name": "GLM5.2-Fast", "description": "The same model served for high TPS.", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 3, "output": 10.25, "cache_read": 0.5, "cache_write": 0 } }, "Kimi-K2.6": { "id": "Kimi-K2.6", "name": "Kimi K2.6", "description": "Kimi K2.6 sparse MoE model with a 262K context window. Available serverless and not included in standard Wafer Pass. Non-ZDR only: requests with `Wafer-ZDR: required` are rejected.", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 1.14, "output": 4.8, "cache_read": 0.19, "cache_write": 0 } }, "MiniMax-M3": { "id": "MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "max"] } ], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 128000 }, "cost": { "input": 0.33, "output": 1.32, "cache_read": 0.07, "cache_write": 0, "tiers": [ { "input": 0.66, "output": 2.64, "cache_read": 0.13, "cache_write": 0, "tier": { "type": "context", "size": 512000 } } ], "context_over_200k": { "input": 0.66, "output": 2.64, "cache_read": 0.13, "cache_write": 0 } } }, "GLM-5.2": { "id": "GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 1.2, "output": 4.1, "cache_read": 0.2, "cache_write": 0 } }, "GLM-5.1": { "id": "GLM-5.1", "name": "GLM-5.1", "description": "General Language Model 5.1 — high-quality bilingual (EN/ZH) generation with strong coding and reasoning capabilities.", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-07", "last_updated": "2026-06-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 202752, "output": 131072 }, "cost": { "input": 1, "output": 3.2, "cache_read": 0.1, "cache_write": 0 } } } }, "minimax": { "id": "minimax", "env": ["MINIMAX_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://api.minimax.io/anthropic/v1", "name": "MiniMax (minimax.io)", "doc": "https://platform.minimax.io/docs/guides/quickstart", "models": { "MiniMax-M2.1": { "id": "MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "Earlier MiniMax agent model for practical coding and productivity tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMax-M2.5-highspeed": { "id": "MiniMax-M2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "MiniMax-M2.7-highspeed": { "id": "MiniMax-M2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375 } }, "MiniMax-M2": { "id": "MiniMax-M2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2 } }, "MiniMax-M2.5": { "id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375 } }, "MiniMax-M3": { "id": "MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "tiers": [ { "input": 0.6, "output": 2.4, "cache_read": 0.12, "tier": { "type": "context", "size": 512000 } } ], "context_over_200k": { "input": 0.6, "output": 2.4, "cache_read": 0.12 } } }, "MiniMax-M2.7": { "id": "MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375 } } } }, "github-copilot": { "id": "github-copilot", "env": ["GITHUB_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.githubcopilot.com", "name": "GitHub Copilot", "doc": "https://docs.github.com/en/copilot", "models": { "claude-sonnet-4.5": { "id": "claude-sonnet-4.5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 168000, "output": 32000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "claude-sonnet-4": { "id": "claude-sonnet-4", "name": "Claude Sonnet 4 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 216000, "input": 128000, "output": 16000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gemini-2.5-pro": { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 128, "max": 32768 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": { "input": ["text", "image", "audio", "video", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 64000 }, "cost": { "input": 1.25, "output": 10, "cache_read": 0.125 } }, "claude-haiku-4.5": { "id": "claude-haiku-4.5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 136000, "output": 64000 }, "cost": { "input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25 } }, "gemini-3.5-flash": { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["minimal", "low", "medium", "high"] }, { "type": "budget_tokens", "min": 256, "max": 24000 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 128000, "output": 64000 }, "cost": { "input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5 } }, "kimi-k2.7-code": { "id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "input": 224000, "output": 32000 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.19 } }, "claude-sonnet-5": { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5 } }, "gpt-5.4-nano": { "id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.2, "output": 1.25, "cache_read": 0.02 } }, "claude-opus-4.7": { "id": "claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 168000, "output": 32000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "mai-code-1-flash-picker": { "id": "mai-code-1-flash-picker", "name": "MAI-Code-1-Flash", "description": "Microsoft coding model built for fast, efficient assistance in everyday developer workflows", "family": "mai", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-06-02", "last_updated": "2026-06-08", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "input": 128000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "gpt-5.2": { "id": "gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.3-codex": { "id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 1, "output": 6, "cache_read": 0.1, "tiers": [ { "input": 2, "output": 9, "cache_read": 0.2, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 2, "output": 9, "cache_read": 0.2 } } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "claude-opus-4.8": { "id": "claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 168000, "output": 64000 }, "experimental": { "modes": { "fast": { "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "claude-fable-5": { "id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "output": 128000 }, "cost": { "input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5 } }, "claude-opus-4.5": { "id": "claude-opus-4.5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 168000, "output": 32000 }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [ { "input": 5, "output": 22.5, "cache_read": 0.5, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 5, "output": 22.5, "cache_read": 0.5 } } }, "gpt-5.4-mini": { "id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 0.75, "output": 4.5, "cache_read": 0.075 } }, "gpt-4.1": { "id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 16384 }, "cost": { "input": 2, "output": 8, "cache_read": 0.5 } }, "gemini-3.1-pro-preview": { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 256, "max": 32000 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 936000, "output": 64000 }, "cost": { "input": 2, "output": 12, "cache_read": 0.2, "tiers": [ { "input": 4, "output": 18, "cache_read": 0.4, "tier": { "type": "context", "size": 200000 } } ], "context_over_200k": { "input": 4, "output": 18, "cache_read": 0.4 } } }, "claude-sonnet-4.6": { "id": "claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] }, { "type": "budget_tokens", "min": 1024, "max": 32000 } ], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 168000, "output": 32000 }, "cost": { "input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75 } }, "gpt-5-mini": { "id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 264000, "input": 128000, "output": 64000 }, "cost": { "input": 0.25, "output": 2, "cache_read": 0.025 } }, "gemini-3-flash-preview": { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] }, { "type": "budget_tokens", "min": 256, "max": 32000 } ], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": { "input": ["text", "image", "video", "audio", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 128000, "output": 64000 }, "cost": { "input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1 } }, "claude-opus-4.6": { "id": "claude-opus-4.6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high", "max"] } ], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "input": 168000, "output": 32000 }, "experimental": { "modes": { "fast": { "cost": { "input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5 }, "provider": { "body": { "speed": "fast" }, "headers": { "anthropic-beta": "fast-mode-2026-02-01" } } } } }, "cost": { "input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25 } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } }, "gpt-5.2-codex": { "id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 400000, "input": 272000, "output": 128000 }, "cost": { "input": 1.75, "output": 14, "cache_read": 0.175 } }, "gpt-5.5": { "id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high", "xhigh"] } ], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": { "input": ["text", "image", "pdf"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1050000, "input": 922000, "output": 128000 }, "cost": { "input": 5, "output": 30, "cache_read": 0.5, "tiers": [ { "input": 10, "output": 45, "cache_read": 1, "tier": { "type": "context", "size": 272000 } } ], "context_over_200k": { "input": 10, "output": 45, "cache_read": 1 } } } } }, "clarifai": { "id": "clarifai", "env": ["CLARIFAI_PAT"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.clarifai.com/v2/ext/openai/v1", "name": "Clarifai", "doc": "https://docs.clarifai.com/compute/inference/", "models": { "moonshotai/chat-completion/models/Kimi-K2_6": { "id": "moonshotai/chat-completion/models/Kimi-K2_6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.95, "output": 4 } }, "minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput": { "id": "minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput", "name": "MiniMax-M2.5 High Throughput", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 204800, "output": 131072 }, "cost": { "input": 0.3, "output": 1.2 } }, "openai/chat-completion/models/gpt-oss-120b-high-throughput": { "id": "openai/chat-completion/models/gpt-oss-120b-high-throughput", "name": "GPT OSS 120B High Throughput", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2026-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.09, "output": 0.36 } }, "openai/chat-completion/models/gpt-oss-20b": { "id": "openai/chat-completion/models/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-12-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 16384 }, "cost": { "input": 0.045, "output": 0.18 } }, "mistralai/completion/models/Ministral-3-14B-Reasoning-2512": { "id": "mistralai/completion/models/Ministral-3-14B-Reasoning-2512", "name": "Ministral 3 14B Reasoning 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-01", "last_updated": "2025-12-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 2.5, "output": 1.7 } }, "mistralai/completion/models/Ministral-3-3B-Reasoning-2512": { "id": "mistralai/completion/models/Ministral-3-3B-Reasoning-2512", "name": "Ministral 3 3B Reasoning 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12", "last_updated": "2026-02-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 1.039, "output": 0.54825 } }, "deepseek-ai/deepseek-ocr/models/DeepSeek-OCR": { "id": "deepseek-ai/deepseek-ocr/models/DeepSeek-OCR", "name": "DeepSeek OCR", "description": "OCR model for extracting structured text from documents and screenshots", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-10-20", "last_updated": "2026-02-25", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0.2, "output": 0.7 } }, "qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507": { "id": "qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507", "name": "Qwen3 30B A3B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-31", "last_updated": "2026-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 131072 }, "cost": { "input": 0.36, "output": 1.3 } }, "qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507": { "id": "qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507", "name": "Qwen3 30B A3B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-30", "last_updated": "2026-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 262144 }, "cost": { "input": 0.3, "output": 0.5 } }, "qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct": { "id": "qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct", "name": "Qwen3 Coder 30B A3B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-31", "last_updated": "2026-02-12", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.11458, "output": 0.74812 } }, "arcee_ai/AFM/models/trinity-mini": { "id": "arcee_ai/AFM/models/trinity-mini", "name": "Trinity Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "trinity-mini", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-12", "last_updated": "2026-02-25", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 131072 }, "cost": { "input": 0.045, "output": 0.15 } }, "clarifai/main/models/mm-poly-8b": { "id": "clarifai/main/models/mm-poly-8b", "name": "MM Poly 8B", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "mm-poly", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-06", "last_updated": "2026-02-25", "modalities": { "input": ["text", "image", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 32768, "output": 4096 }, "cost": { "input": 0.658, "output": 1.11 } } } }, "the-grid-ai": { "id": "the-grid-ai", "env": ["THEGRIDAI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.thegrid.ai/v1", "name": "The Grid AI", "doc": "https://thegrid.ai/docs", "models": { "agent-prime": { "id": "agent-prime", "name": "Agent Prime", "description": "Reliable models for dependable agentic applications, multi-step tool use, and reasoning workflows. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-04", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 196608, "input": 120000, "output": 30000 }, "status": "beta" }, "agent-max": { "id": "agent-max", "name": "Agent Max", "description": "Frontier models for autonomous research, deep multi-step tool chains, and complex long-horizon tasks. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-04", "last_updated": "2026-07-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 }, "status": "beta" }, "text-standard": { "id": "text-standard", "name": "Text Standard", "description": "Price-optimized models with low-latency, high-throughput and shorter maximum outputs. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 120000, "output": 16000 } }, "code-prime": { "id": "code-prime", "name": "Code Prime", "description": "Reliable models for everyday software tasks, code completion, review, and standard debugging. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-04", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 196608, "input": 120000, "output": 30000 }, "status": "beta" }, "text-prime": { "id": "text-prime", "name": "Text Prime", "description": "Reliable models for everyday text generation, editing, and analysis across diverse workflows. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 196608, "input": 120000, "output": 30000 } }, "code-max": { "id": "code-max", "name": "Code Max", "description": "Frontier models for complex research, architectural decisions, debugging, and multi-file development. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-04", "last_updated": "2026-07-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 }, "status": "beta" }, "agent-standard": { "id": "agent-standard", "name": "Agent Standard", "description": "Price-optimized models for fast tool calls, simple agent loops, high-throughput automation, and orchestration. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-04", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 120000, "output": 16000 }, "status": "beta" }, "text-max": { "id": "text-max", "name": "Text Max", "description": "Frontier models for deep reasoning, long context, and complex workflows. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" }, { "type": "effort", "values": ["low", "medium", "high", "xhigh", "max"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-03-24", "last_updated": "2026-07-06", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1000000, "input": 922000, "output": 128000 } }, "code-standard": { "id": "code-standard", "name": "Code Standard", "description": "Price-optimized models for rapid autocomplete, linting, high-frequency suggestions, and batch edits. Any model that meets the contract spec can serve your request.", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-05-04", "last_updated": "2026-07-06", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "input": 120000, "output": 16000 }, "status": "beta" } } }, "synthetic": { "id": "synthetic", "env": ["SYNTHETIC_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.synthetic.new/openai/v1", "name": "Synthetic", "doc": "https://synthetic.new/pricing", "models": { "hf:moonshotai/Kimi-K2.7-Code": { "id": "hf:moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.95, "output": 4, "cache_read": 0.95 } }, "hf:zai-org/GLM-4.7-Flash": { "id": "hf:zai-org/GLM-4.7-Flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 196608, "output": 65536 }, "cost": { "input": 0.1, "output": 0.5, "cache_read": 0.1 } }, "hf:zai-org/GLM-5.2": { "id": "hf:zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "high", "xhigh"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 65536 }, "cost": { "input": 1.4, "output": 4.4, "cache_read": 1.4 } }, "hf:MiniMaxAI/MiniMax-M3": { "id": "hf:MiniMaxAI/MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 524288, "output": 65536 }, "cost": { "input": 0.6, "output": 1.2, "cache_read": 0.6 } }, "hf:openai/gpt-oss-120b": { "id": "hf:openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 131072, "output": 32768 }, "cost": { "input": 0.1, "output": 0.1, "cache_read": 0.1 } }, "hf:Qwen/Qwen3.6-27B": { "id": "hf:Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["none", "low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.45, "output": 3.6, "cache_read": 0.45 } }, "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": { "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", "name": "Nemotron 3 Super 120B A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "effort", "values": ["low", "medium", "high"] } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 262144, "output": 65536 }, "cost": { "input": 0.3, "output": 1, "cache_read": 0.3 } } } }, "iflowcn": { "id": "iflowcn", "env": ["IFLOW_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://apis.iflow.cn/v1", "name": "iFlow", "doc": "https://platform.iflow.cn/en/docs", "models": { "qwen3-coder-plus": { "id": "qwen3-coder-plus", "name": "Qwen3-Coder-Plus", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "deepseek-v3": { "id": "deepseek-v3", "name": "DeepSeek-V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-26", "last_updated": "2024-12-26", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "kimi-k2": { "id": "kimi-k2", "name": "Kimi-K2", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-32b": { "id": "qwen3-32b", "name": "Qwen3-32B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-max-preview": { "id": "qwen3-max-preview", "name": "Qwen3-Max-Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-max": { "id": "qwen3-max", "name": "Qwen3-Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-235b": { "id": "qwen3-235b", "name": "Qwen3-235B-A22B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-01", "last_updated": "2024-12-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "glm-4.6": { "id": "glm-4.6", "name": "GLM-4.6", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-12-01", "last_updated": "2025-11-13", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 200000, "output": 128000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-235b-a22b-thinking-2507": { "id": "qwen3-235b-a22b-thinking-2507", "name": "Qwen3-235B-A22B-Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "deepseek-r1": { "id": "deepseek-r1", "name": "DeepSeek-R1", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-20", "last_updated": "2025-01-20", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-vl-plus": { "id": "qwen3-vl-plus", "name": "Qwen3-VL-Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text", "image"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 32000 }, "cost": { "input": 0, "output": 0 } }, "qwen3-235b-a22b-instruct": { "id": "qwen3-235b-a22b-instruct", "name": "Qwen3-235B-A22B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-01", "last_updated": "2025-07-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "kimi-k2-0905": { "id": "kimi-k2-0905", "name": "Kimi-K2-0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0, "output": 0 } }, "deepseek-v3.2": { "id": "deepseek-v3.2", "name": "DeepSeek-V3.2-Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 128000, "output": 64000 }, "cost": { "input": 0, "output": 0 } } } }, "xiaomi-token-plan-sgp": { "id": "xiaomi-token-plan-sgp", "env": ["XIAOMI_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://token-plan-sgp.xiaomimimo.com/v1", "name": "Xiaomi Token Plan (Singapore)", "doc": "https://platform.xiaomimimo.com/#/docs", "models": { "mimo-v2.5-tts": { "id": "mimo-v2.5-tts", "name": "MiMo-V2.5-TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5-pro": { "id": "mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2-pro": { "id": "mimo-v2-pro", "name": "MiMo-V2-Pro", "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["text"] }, "open_weights": false, "limit": { "context": 1048576, "output": 131072 }, "status": "deprecated", "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2-tts": { "id": "mimo-v2-tts", "name": "MiMo-V2-TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5": { "id": "mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [ { "type": "toggle" } ], "tool_call": true, "interleaved": { "field": "reasoning_content" }, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": true, "limit": { "context": 1048576, "output": 131072 }, "cost": { "input": 0, "output": 0, "cache_read": 0 } }, "mimo-v2.5-tts-voicedesign": { "id": "mimo-v2.5-tts-voicedesign", "name": "MiMo-V2.5-TTS-VoiceDesign", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } }, "mimo-v2.5-tts-voiceclone": { "id": "mimo-v2.5-tts-voiceclone", "name": "MiMo-V2.5-TTS-VoiceClone", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "mimo", "attachment": false, "reasoning": false, "tool_call": false, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": { "input": ["text"], "output": ["audio"] }, "open_weights": true, "limit": { "context": 8192, "output": 8192 }, "cost": { "input": 0, "output": 0 } } } }, "claudinio": { "id": "claudinio", "env": ["CLAUDINIO_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.claudin.io/v1", "name": "Claudinio", "doc": "https://claudin.io", "models": { "claudinio": { "id": "claudinio", "name": "Claudinio", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "knowledge": "2026-05", "release_date": "2026-05-12", "last_updated": "2026-06-02", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 0.5, "output": 2, "cache_read": 0.15 } }, "claudius": { "id": "claudius", "name": "Claudius", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "knowledge": "2026-05", "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": { "input": ["text", "image", "audio", "video"], "output": ["text"] }, "open_weights": false, "limit": { "context": 256000, "output": 64000 }, "cost": { "input": 3, "output": 8, "cache_read": 0.9 } } } } }