Merge main into model-registry

This commit is contained in:
Mario Zechner
2026-06-22 14:00:18 +02:00
220 changed files with 10488 additions and 4354 deletions
+5 -22
View File
@@ -13,30 +13,13 @@ export const CEREBRAS_MODELS = {
reasoning: true,
input: ["text"],
cost: {
input: 0.25,
output: 0.69,
input: 0.35,
output: 0.75,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 131072,
maxTokens: 32768,
} satisfies Model<"openai-completions">,
"llama3.1-8b": {
id: "llama3.1-8b",
name: "Llama 3.1 8B",
api: "openai-completions",
provider: "cerebras",
baseUrl: "https://api.cerebras.ai/v1",
reasoning: false,
input: ["text"],
cost: {
input: 0.1,
output: 0.1,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 32000,
maxTokens: 8000,
maxTokens: 40960,
} satisfies Model<"openai-completions">,
"zai-glm-4.7": {
id: "zai-glm-4.7",
@@ -44,7 +27,7 @@ export const CEREBRAS_MODELS = {
api: "openai-completions",
provider: "cerebras",
baseUrl: "https://api.cerebras.ai/v1",
reasoning: false,
reasoning: true,
input: ["text"],
cost: {
input: 2.25,
@@ -53,6 +36,6 @@ export const CEREBRAS_MODELS = {
cacheWrite: 0,
},
contextWindow: 131072,
maxTokens: 40000,
maxTokens: 40960,
} satisfies Model<"openai-completions">,
} as const;
@@ -112,6 +112,24 @@ export const CLOUDFLARE_WORKERS_AI_MODELS = {
contextWindow: 262144,
maxTokens: 256000,
} satisfies Model<"openai-completions">,
"@cf/moonshotai/kimi-k2.7-code": {
id: "@cf/moonshotai/kimi-k2.7-code",
name: "Kimi K2.7 Code",
api: "openai-completions",
provider: "cloudflare-workers-ai",
baseUrl: "https://api.cloudflare.com/client/v4/accounts/{CLOUDFLARE_ACCOUNT_ID}/ai/v1",
compat: {"sendSessionAffinityHeaders":true},
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.95,
output: 4,
cacheRead: 0.19,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"@cf/nvidia/nemotron-3-120b-a12b": {
id: "@cf/nvidia/nemotron-3-120b-a12b",
name: "Nemotron 3 Super 120B",
@@ -202,4 +220,22 @@ export const CLOUDFLARE_WORKERS_AI_MODELS = {
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"@cf/zai-org/glm-5.2": {
id: "@cf/zai-org/glm-5.2",
name: "Glm 5.2",
api: "openai-completions",
provider: "cloudflare-workers-ai",
baseUrl: "https://api.cloudflare.com/client/v4/accounts/{CLOUDFLARE_ACCOUNT_ID}/ai/v1",
compat: {"sendSessionAffinityHeaders":true},
reasoning: true,
input: ["text"],
cost: {
input: 1.4,
output: 4.4,
cacheRead: 0.26,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
} as const;
+71 -34
View File
@@ -16,7 +16,7 @@ export const FIREWORKS_MODELS = {
cost: {
input: 0.14,
output: 0.28,
cacheRead: 0.03,
cacheRead: 0.028,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -58,6 +58,25 @@ export const FIREWORKS_MODELS = {
contextWindow: 202800,
maxTokens: 131072,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/models/glm-5p2": {
id: "accounts/fireworks/models/glm-5p2",
name: "GLM 5.2",
api: "openai-completions",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false},
reasoning: true,
thinkingLevelMap: {"off":"none","minimal":null,"low":"high","medium":"high","xhigh":"max"},
input: ["text"],
cost: {
input: 1.4,
output: 4.4,
cacheRead: 0.26,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"accounts/fireworks/models/gpt-oss-120b": {
id: "accounts/fireworks/models/gpt-oss-120b",
name: "GPT OSS 120B",
@@ -94,24 +113,6 @@ export const FIREWORKS_MODELS = {
contextWindow: 131072,
maxTokens: 32768,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/models/kimi-k2p5": {
id: "accounts/fireworks/models/kimi-k2p5",
name: "Kimi K2.5",
api: "anthropic-messages",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference",
compat: {"sendSessionAffinityHeaders":true,"supportsEagerToolInputStreaming":false,"supportsCacheControlOnTools":false,"supportsLongCacheRetention":false},
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.6,
output: 3,
cacheRead: 0.1,
cacheWrite: 0,
},
contextWindow: 256000,
maxTokens: 256000,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/models/kimi-k2p6": {
id: "accounts/fireworks/models/kimi-k2p6",
name: "Kimi K2.6",
@@ -130,23 +131,23 @@ export const FIREWORKS_MODELS = {
contextWindow: 262000,
maxTokens: 262000,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/models/minimax-m2p5": {
id: "accounts/fireworks/models/minimax-m2p5",
name: "MiniMax-M2.5",
"accounts/fireworks/models/kimi-k2p7-code": {
id: "accounts/fireworks/models/kimi-k2p7-code",
name: "Kimi K2.7 Code",
api: "anthropic-messages",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference",
compat: {"sendSessionAffinityHeaders":true,"supportsEagerToolInputStreaming":false,"supportsCacheControlOnTools":false,"supportsLongCacheRetention":false},
reasoning: true,
input: ["text"],
input: ["text", "image"],
cost: {
input: 0.3,
output: 1.2,
cacheRead: 0.03,
input: 0.95,
output: 4,
cacheRead: 0.19,
cacheWrite: 0,
},
contextWindow: 196608,
maxTokens: 196608,
contextWindow: 262000,
maxTokens: 262000,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/models/minimax-m2p7": {
id: "accounts/fireworks/models/minimax-m2p7",
@@ -166,9 +167,27 @@ export const FIREWORKS_MODELS = {
contextWindow: 196608,
maxTokens: 196608,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/models/qwen3p6-plus": {
id: "accounts/fireworks/models/qwen3p6-plus",
name: "Qwen 3.6 Plus",
"accounts/fireworks/models/minimax-m3": {
id: "accounts/fireworks/models/minimax-m3",
name: "MiniMax-M3",
api: "anthropic-messages",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference",
compat: {"sendSessionAffinityHeaders":true,"supportsEagerToolInputStreaming":false,"supportsCacheControlOnTools":false,"supportsLongCacheRetention":false},
reasoning: true,
input: ["text"],
cost: {
input: 0.3,
output: 1.2,
cacheRead: 0.06,
cacheWrite: 0,
},
contextWindow: 512000,
maxTokens: 512000,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/models/qwen3p7-plus": {
id: "accounts/fireworks/models/qwen3p7-plus",
name: "Qwen 3.7 Plus",
api: "anthropic-messages",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference",
@@ -176,9 +195,9 @@ export const FIREWORKS_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.5,
output: 3,
cacheRead: 0.1,
input: 0.4,
output: 1.6,
cacheRead: 0.08,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -238,4 +257,22 @@ export const FIREWORKS_MODELS = {
contextWindow: 262000,
maxTokens: 262000,
} satisfies Model<"anthropic-messages">,
"accounts/fireworks/routers/kimi-k2p7-code-fast": {
id: "accounts/fireworks/routers/kimi-k2p7-code-fast",
name: "Kimi K2.7 Code Fast",
api: "anthropic-messages",
provider: "fireworks",
baseUrl: "https://api.fireworks.ai/inference",
compat: {"sendSessionAffinityHeaders":true,"supportsEagerToolInputStreaming":false,"supportsCacheControlOnTools":false,"supportsLongCacheRetention":false},
reasoning: true,
input: ["text", "image"],
cost: {
input: 1.9,
output: 8,
cacheRead: 0.38,
cacheWrite: 0,
},
contextWindow: 262000,
maxTokens: 262000,
} satisfies Model<"anthropic-messages">,
} as const;
+6 -2
View File
@@ -1,15 +1,19 @@
import { anthropicMessagesApi } from "../api/anthropic-messages.lazy.ts";
import { openAICompletionsApi } from "../api/openai-completions.lazy.ts";
import { envApiKeyAuth } from "../auth/helpers.ts";
import { createProvider, type Provider } from "../models.ts";
import { FIREWORKS_MODELS } from "./fireworks.models.ts";
export function fireworksProvider(): Provider<"anthropic-messages"> {
export function fireworksProvider(): Provider<"anthropic-messages" | "openai-completions"> {
return createProvider({
id: "fireworks",
name: "Fireworks",
baseUrl: "https://api.fireworks.ai/inference",
auth: { apiKey: envApiKeyAuth("Fireworks API key", ["FIREWORKS_API_KEY"]) },
models: Object.values(FIREWORKS_MODELS),
api: anthropicMessagesApi(),
api: {
"anthropic-messages": anthropicMessagesApi(),
"openai-completions": openAICompletionsApi(),
},
});
}
@@ -4,6 +4,25 @@
import type { Model } from "../types.ts";
export const GITHUB_COPILOT_MODELS = {
"claude-fable-5": {
id: "claude-fable-5",
name: "Claude Fable 5",
api: "openai-completions",
provider: "github-copilot",
baseUrl: "https://api.individual.githubcopilot.com",
headers: {"User-Agent":"GitHubCopilotChat/0.35.0","Editor-Version":"vscode/1.107.0","Editor-Plugin-Version":"copilot-chat/0.35.0","Copilot-Integration-Id":"vscode-chat"},
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false},
reasoning: true,
input: ["text", "image"],
cost: {
input: 10,
output: 50,
cacheRead: 1,
cacheWrite: 12.5,
},
contextWindow: 1000000,
maxTokens: 128000,
} satisfies Model<"openai-completions">,
"claude-haiku-4.5": {
id: "claude-haiku-4.5",
name: "Claude Haiku 4.5 (latest)",
@@ -70,7 +89,7 @@ export const GITHUB_COPILOT_MODELS = {
headers: {"User-Agent":"GitHubCopilotChat/0.35.0","Editor-Version":"vscode/1.107.0","Editor-Plugin-Version":"copilot-chat/0.35.0","Copilot-Integration-Id":"vscode-chat"},
compat: {"forceAdaptiveThinking":true,"supportsTemperature":false},
reasoning: true,
thinkingLevelMap: {"xhigh":"xhigh"},
thinkingLevelMap: {"xhigh":"xhigh","minimal":"low"},
input: ["text", "image"],
cost: {
input: 5,
@@ -90,7 +109,7 @@ export const GITHUB_COPILOT_MODELS = {
headers: {"User-Agent":"GitHubCopilotChat/0.35.0","Editor-Version":"vscode/1.107.0","Editor-Plugin-Version":"copilot-chat/0.35.0","Copilot-Integration-Id":"vscode-chat"},
compat: {"forceAdaptiveThinking":true,"supportsTemperature":false},
reasoning: true,
thinkingLevelMap: {"xhigh":"xhigh"},
thinkingLevelMap: {"xhigh":"xhigh","minimal":"low"},
input: ["text", "image"],
cost: {
input: 5,
@@ -148,6 +167,7 @@ export const GITHUB_COPILOT_MODELS = {
headers: {"User-Agent":"GitHubCopilotChat/0.35.0","Editor-Version":"vscode/1.107.0","Editor-Plugin-Version":"copilot-chat/0.35.0","Copilot-Integration-Id":"vscode-chat"},
compat: {"forceAdaptiveThinking":true},
reasoning: true,
thinkingLevelMap: {"minimal":"low","xhigh":"max"},
input: ["text", "image"],
cost: {
input: 3,
@@ -405,23 +425,4 @@ export const GITHUB_COPILOT_MODELS = {
contextWindow: 400000,
maxTokens: 128000,
} satisfies Model<"openai-responses">,
"raptor-mini": {
id: "raptor-mini",
name: "Raptor mini",
api: "openai-completions",
provider: "github-copilot",
baseUrl: "https://api.individual.githubcopilot.com",
headers: {"User-Agent":"GitHubCopilotChat/0.35.0","Editor-Version":"vscode/1.107.0","Editor-Plugin-Version":"copilot-chat/0.35.0","Copilot-Integration-Id":"vscode-chat"},
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false},
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.25,
output: 2,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 400000,
maxTokens: 128000,
} satisfies Model<"openai-completions">,
} as const;
+69 -117
View File
@@ -4,94 +4,9 @@
import type { Model } from "../types.ts";
export const GOOGLE_VERTEX_MODELS = {
"gemini-1.5-flash": {
id: "gemini-1.5-flash",
name: "Gemini 1.5 Flash (Vertex)",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.075,
output: 0.3,
cacheRead: 0.01875,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 8192,
} satisfies Model<"google-vertex">,
"gemini-1.5-flash-8b": {
id: "gemini-1.5-flash-8b",
name: "Gemini 1.5 Flash-8B (Vertex)",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.0375,
output: 0.15,
cacheRead: 0.01,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 8192,
} satisfies Model<"google-vertex">,
"gemini-1.5-pro": {
id: "gemini-1.5-pro",
name: "Gemini 1.5 Pro (Vertex)",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: false,
input: ["text", "image"],
cost: {
input: 1.25,
output: 5,
cacheRead: 0.3125,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 8192,
} satisfies Model<"google-vertex">,
"gemini-2.0-flash": {
id: "gemini-2.0-flash",
name: "Gemini 2.0 Flash (Vertex)",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.15,
output: 0.6,
cacheRead: 0.0375,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 8192,
} satisfies Model<"google-vertex">,
"gemini-2.0-flash-lite": {
id: "gemini-2.0-flash-lite",
name: "Gemini 2.0 Flash Lite (Vertex)",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.075,
output: 0.3,
cacheRead: 0.01875,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
"gemini-2.5-flash": {
id: "gemini-2.5-flash",
name: "Gemini 2.5 Flash (Vertex)",
name: "Gemini 2.5 Flash",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
@@ -108,24 +23,7 @@ export const GOOGLE_VERTEX_MODELS = {
} satisfies Model<"google-vertex">,
"gemini-2.5-flash-lite": {
id: "gemini-2.5-flash-lite",
name: "Gemini 2.5 Flash Lite (Vertex)",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.1,
output: 0.4,
cacheRead: 0.01,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
"gemini-2.5-flash-lite-preview-09-2025": {
id: "gemini-2.5-flash-lite-preview-09-2025",
name: "Gemini 2.5 Flash Lite Preview 09-25 (Vertex)",
name: "Gemini 2.5 Flash-Lite",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
@@ -142,7 +40,7 @@ export const GOOGLE_VERTEX_MODELS = {
} satisfies Model<"google-vertex">,
"gemini-2.5-pro": {
id: "gemini-2.5-pro",
name: "Gemini 2.5 Pro (Vertex)",
name: "Gemini 2.5 Pro",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
@@ -159,7 +57,7 @@ export const GOOGLE_VERTEX_MODELS = {
} satisfies Model<"google-vertex">,
"gemini-3-flash-preview": {
id: "gemini-3-flash-preview",
name: "Gemini 3 Flash Preview (Vertex)",
name: "Gemini 3 Flash Preview",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
@@ -175,27 +73,27 @@ export const GOOGLE_VERTEX_MODELS = {
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
"gemini-3-pro-preview": {
id: "gemini-3-pro-preview",
name: "Gemini 3 Pro Preview (Vertex)",
"gemini-3.1-flash-lite": {
id: "gemini-3.1-flash-lite",
name: "Gemini 3.1 Flash Lite",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: true,
thinkingLevelMap: {"off":null,"minimal":null,"low":"LOW","medium":null,"high":"HIGH"},
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 2,
output: 12,
cacheRead: 0.2,
input: 0.25,
output: 1.5,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 64000,
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
"gemini-3.1-pro-preview": {
id: "gemini-3.1-pro-preview",
name: "Gemini 3.1 Pro Preview (Vertex)",
name: "Gemini 3.1 Pro Preview",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
@@ -213,7 +111,7 @@ export const GOOGLE_VERTEX_MODELS = {
} satisfies Model<"google-vertex">,
"gemini-3.1-pro-preview-customtools": {
id: "gemini-3.1-pro-preview-customtools",
name: "Gemini 3.1 Pro Preview Custom Tools (Vertex)",
name: "Gemini 3.1 Pro Preview Custom Tools",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
@@ -229,4 +127,58 @@ export const GOOGLE_VERTEX_MODELS = {
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
"gemini-3.5-flash": {
id: "gemini-3.5-flash",
name: "Gemini 3.5 Flash",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 1.5,
output: 9,
cacheRead: 0.15,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
"gemini-flash-latest": {
id: "gemini-flash-latest",
name: "Gemini Flash Latest",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 1.5,
output: 9,
cacheRead: 0.15,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
"gemini-flash-lite-latest": {
id: "gemini-flash-lite-latest",
name: "Gemini Flash-Lite Latest",
api: "google-vertex",
provider: "google-vertex",
baseUrl: "https://{location}-aiplatform.googleapis.com",
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 0.25,
output: 1.5,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 1048576,
maxTokens: 65536,
} satisfies Model<"google-vertex">,
} as const;
+7 -5
View File
@@ -222,11 +222,12 @@ export const GOOGLE_MODELS = {
provider: "google",
baseUrl: "https://generativelanguage.googleapis.com/v1beta",
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 0.3,
output: 2.5,
cacheRead: 0.075,
input: 1.5,
output: 9,
cacheRead: 0.15,
cacheWrite: 0,
},
contextWindow: 1048576,
@@ -239,10 +240,11 @@ export const GOOGLE_MODELS = {
provider: "google",
baseUrl: "https://generativelanguage.googleapis.com/v1beta",
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 0.1,
output: 0.4,
input: 0.25,
output: 1.5,
cacheRead: 0.025,
cacheWrite: 0,
},
@@ -4,6 +4,24 @@
import type { Model } from "../types.ts";
export const KIMI_CODING_MODELS = {
"k2p7": {
id: "k2p7",
name: "Kimi K2.7 Code",
api: "anthropic-messages",
provider: "kimi-coding",
baseUrl: "https://api.kimi.com/coding",
headers: {"User-Agent":"KimiCLI/1.5"},
reasoning: true,
input: ["text", "image"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 32768,
} satisfies Model<"anthropic-messages">,
"kimi-for-coding": {
id: "kimi-for-coding",
name: "Kimi For Coding",
+28 -28
View File
@@ -15,7 +15,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.3,
output: 0.9,
cacheRead: 0,
cacheRead: 0.03,
cacheWrite: 0,
},
contextWindow: 256000,
@@ -32,7 +32,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.4,
output: 2,
cacheRead: 0,
cacheRead: 0.04,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -49,7 +49,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.4,
output: 2,
cacheRead: 0,
cacheRead: 0.04,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -66,7 +66,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.4,
output: 2,
cacheRead: 0,
cacheRead: 0.04,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -83,7 +83,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.4,
output: 2,
cacheRead: 0,
cacheRead: 0.04,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -100,7 +100,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.1,
output: 0.3,
cacheRead: 0,
cacheRead: 0.01,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -117,7 +117,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.1,
output: 0.3,
cacheRead: 0,
cacheRead: 0.01,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -151,7 +151,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 2,
output: 5,
cacheRead: 0,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -168,7 +168,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.5,
output: 1.5,
cacheRead: 0,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -185,7 +185,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.04,
output: 0.04,
cacheRead: 0,
cacheRead: 0.004,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -202,7 +202,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.1,
output: 0.1,
cacheRead: 0,
cacheRead: 0.01,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -219,7 +219,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 2,
output: 6,
cacheRead: 0,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 131072,
@@ -236,7 +236,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.5,
output: 1.5,
cacheRead: 0,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -253,7 +253,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.5,
output: 1.5,
cacheRead: 0,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -270,7 +270,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.4,
output: 2,
cacheRead: 0,
cacheRead: 0.04,
cacheWrite: 0,
},
contextWindow: 131072,
@@ -287,7 +287,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.4,
output: 2,
cacheRead: 0,
cacheRead: 0.04,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -304,7 +304,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 1.5,
output: 7.5,
cacheRead: 0,
cacheRead: 0.15,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -338,7 +338,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.4,
output: 2,
cacheRead: 0,
cacheRead: 0.04,
cacheWrite: 0,
},
contextWindow: 262144,
@@ -355,7 +355,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.15,
output: 0.15,
cacheRead: 0,
cacheRead: 0.015,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -372,7 +372,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.1,
output: 0.3,
cacheRead: 0,
cacheRead: 0.01,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -389,7 +389,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.15,
output: 0.6,
cacheRead: 0,
cacheRead: 0.015,
cacheWrite: 0,
},
contextWindow: 256000,
@@ -406,7 +406,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.15,
output: 0.6,
cacheRead: 0,
cacheRead: 0.015,
cacheWrite: 0,
},
contextWindow: 256000,
@@ -423,7 +423,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.25,
output: 0.25,
cacheRead: 0,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 8000,
@@ -440,7 +440,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.15,
output: 0.15,
cacheRead: 0,
cacheRead: 0.015,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -457,7 +457,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 2,
output: 6,
cacheRead: 0,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 64000,
@@ -474,7 +474,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.7,
output: 0.7,
cacheRead: 0,
cacheRead: 0.07,
cacheWrite: 0,
},
contextWindow: 32000,
@@ -491,7 +491,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 0.15,
output: 0.15,
cacheRead: 0,
cacheRead: 0.015,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -508,7 +508,7 @@ export const MISTRAL_MODELS = {
cost: {
input: 2,
output: 6,
cacheRead: 0,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -130,4 +130,42 @@ export const MOONSHOTAI_CN_MODELS = {
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"kimi-k2.7-code": {
id: "kimi-k2.7-code",
name: "Kimi K2.7 Code",
api: "openai-completions",
provider: "moonshotai-cn",
baseUrl: "https://api.moonshot.cn/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"thinkingFormat":"deepseek"},
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 0.95,
output: 4,
cacheRead: 0.19,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"kimi-k2.7-code-highspeed": {
id: "kimi-k2.7-code-highspeed",
name: "Kimi K2.7 Code HighSpeed",
api: "openai-completions",
provider: "moonshotai-cn",
baseUrl: "https://api.moonshot.cn/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"thinkingFormat":"deepseek"},
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 1.9,
output: 8,
cacheRead: 0.38,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
} as const;
@@ -130,4 +130,42 @@ export const MOONSHOTAI_MODELS = {
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"kimi-k2.7-code": {
id: "kimi-k2.7-code",
name: "Kimi K2.7 Code",
api: "openai-completions",
provider: "moonshotai",
baseUrl: "https://api.moonshot.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"thinkingFormat":"deepseek"},
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 0.95,
output: 4,
cacheRead: 0.19,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"kimi-k2.7-code-highspeed": {
id: "kimi-k2.7-code-highspeed",
name: "Kimi K2.7 Code HighSpeed",
api: "openai-completions",
provider: "moonshotai",
baseUrl: "https://api.moonshot.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"thinkingFormat":"deepseek"},
reasoning: true,
thinkingLevelMap: {"off":null},
input: ["text", "image"],
cost: {
input: 1.9,
output: 8,
cacheRead: 0.38,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
} as const;
@@ -289,25 +289,6 @@ export const NVIDIA_MODELS = {
contextWindow: 131072,
maxTokens: 32768,
} satisfies Model<"openai-completions">,
"qwen/qwen3-coder-480b-a35b-instruct": {
id: "qwen/qwen3-coder-480b-a35b-instruct",
name: "Qwen3 Coder 480B A35B Instruct",
api: "openai-completions",
provider: "nvidia",
baseUrl: "https://integrate.api.nvidia.com/v1",
headers: {"NVCF-POLL-SECONDS":"3600"},
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false},
reasoning: false,
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 66536,
} satisfies Model<"openai-completions">,
"qwen/qwen3.5-122b-a10b": {
id: "qwen/qwen3.5-122b-a10b",
name: "Qwen3.5 122B-A10B",
+32 -49
View File
@@ -42,24 +42,6 @@ export const OPENCODE_GO_MODELS = {
contextWindow: 1000000,
maxTokens: 384000,
} satisfies Model<"openai-completions">,
"glm-5": {
id: "glm-5",
name: "GLM-5",
api: "openai-completions",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go/v1",
compat: {"maxTokensField":"max_tokens"},
reasoning: true,
input: ["text"],
cost: {
input: 1,
output: 3.2,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 202752,
maxTokens: 32768,
} satisfies Model<"openai-completions">,
"glm-5.1": {
id: "glm-5.1",
name: "GLM-5.1",
@@ -78,23 +60,23 @@ export const OPENCODE_GO_MODELS = {
contextWindow: 202752,
maxTokens: 32768,
} satisfies Model<"openai-completions">,
"kimi-k2.5": {
id: "kimi-k2.5",
name: "Kimi K2.5",
"glm-5.2": {
id: "glm-5.2",
name: "GLM-5.2",
api: "openai-completions",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go/v1",
compat: {"maxTokensField":"max_tokens"},
reasoning: true,
input: ["text", "image"],
input: ["text"],
cost: {
input: 0.6,
output: 3,
cacheRead: 0.1,
input: 1.4,
output: 4.4,
cacheRead: 0.26,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 65536,
contextWindow: 1000000,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"kimi-k2.6": {
id: "kimi-k2.6",
@@ -102,7 +84,7 @@ export const OPENCODE_GO_MODELS = {
api: "openai-completions",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go/v1",
compat: {"thinkingFormat":"deepseek","supportsReasoningEffort":false,"maxTokensField":"max_tokens"},
compat: {"thinkingFormat":"deepseek","supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsLongCacheRetention":false},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text", "image"],
@@ -115,6 +97,24 @@ export const OPENCODE_GO_MODELS = {
contextWindow: 262144,
maxTokens: 65536,
} satisfies Model<"openai-completions">,
"kimi-k2.7-code": {
id: "kimi-k2.7-code",
name: "Kimi K2.7 Code",
api: "openai-completions",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go/v1",
compat: {"maxTokensField":"max_tokens"},
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.95,
output: 4,
cacheRead: 0.19,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"mimo-v2.5": {
id: "mimo-v2.5",
name: "MiMo V2.5",
@@ -151,23 +151,6 @@ export const OPENCODE_GO_MODELS = {
contextWindow: 1048576,
maxTokens: 128000,
} satisfies Model<"openai-completions">,
"minimax-m2.5": {
id: "minimax-m2.5",
name: "MiniMax M2.5",
api: "anthropic-messages",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go",
reasoning: true,
input: ["text"],
cost: {
input: 0.3,
output: 1.2,
cacheRead: 0.03,
cacheWrite: 0,
},
contextWindow: 204800,
maxTokens: 65536,
} satisfies Model<"anthropic-messages">,
"minimax-m2.7": {
id: "minimax-m2.7",
name: "MiniMax M2.7",
@@ -188,16 +171,16 @@ export const OPENCODE_GO_MODELS = {
} satisfies Model<"openai-completions">,
"minimax-m3": {
id: "minimax-m3",
name: "MiniMax M3",
name: "MiniMax M3 (3x usage)",
api: "anthropic-messages",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go",
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.3,
output: 1.2,
cacheRead: 0.06,
input: 0.1,
output: 0.4,
cacheRead: 0.02,
cacheWrite: 0,
},
contextWindow: 512000,
+6 -25
View File
@@ -22,25 +22,6 @@ export const OPENCODE_MODELS = {
contextWindow: 200000,
maxTokens: 32000,
} satisfies Model<"openai-completions">,
"claude-fable-5": {
id: "claude-fable-5",
name: "Claude Fable 5",
api: "anthropic-messages",
provider: "opencode",
baseUrl: "https://opencode.ai/zen",
compat: {"forceAdaptiveThinking":true},
reasoning: true,
thinkingLevelMap: {"off":null,"xhigh":"xhigh"},
input: ["text", "image"],
cost: {
input: 10,
output: 50,
cacheRead: 1,
cacheWrite: 12.5,
},
contextWindow: 1000000,
maxTokens: 128000,
} satisfies Model<"anthropic-messages">,
"claude-haiku-4-5": {
id: "claude-haiku-4-5",
name: "Claude Haiku 4.5",
@@ -207,7 +188,7 @@ export const OPENCODE_MODELS = {
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
compat: {"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek"},
compat: {"maxTokensField":"max_tokens","supportsLongCacheRetention":false,"requiresReasoningContentOnAssistantMessages":true},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"max"},
input: ["text"],
@@ -226,7 +207,7 @@ export const OPENCODE_MODELS = {
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
compat: {"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek"},
compat: {"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"max"},
input: ["text"],
@@ -245,7 +226,7 @@ export const OPENCODE_MODELS = {
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
compat: {"maxTokensField":"max_tokens","requiresReasoningContentOnAssistantMessages":true,"thinkingFormat":"deepseek"},
compat: {"maxTokensField":"max_tokens","supportsLongCacheRetention":false,"requiresReasoningContentOnAssistantMessages":true},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":"max"},
input: ["text"],
@@ -661,7 +642,7 @@ export const OPENCODE_MODELS = {
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
compat: {"maxTokensField":"max_tokens"},
compat: {"maxTokensField":"max_tokens","supportsLongCacheRetention":false},
reasoning: true,
input: ["text", "image"],
cost: {
@@ -679,7 +660,7 @@ export const OPENCODE_MODELS = {
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
compat: {"thinkingFormat":"deepseek","supportsReasoningEffort":false,"maxTokensField":"max_tokens"},
compat: {"thinkingFormat":"deepseek","supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsLongCacheRetention":false},
reasoning: true,
input: ["text", "image"],
cost: {
@@ -733,7 +714,7 @@ export const OPENCODE_MODELS = {
api: "openai-completions",
provider: "opencode",
baseUrl: "https://opencode.ai/zen/v1",
compat: {"maxTokensField":"max_tokens"},
compat: {"maxTokensField":"max_tokens","supportsLongCacheRetention":false},
reasoning: true,
input: ["text"],
cost: {
File diff suppressed because it is too large Load Diff
+107 -109
View File
@@ -4,25 +4,6 @@
import type { Model } from "../types.ts";
export const TOGETHER_MODELS = {
"MiniMaxAI/MiniMax-M2.5": {
id: "MiniMaxAI/MiniMax-M2.5",
name: "MiniMax-M2.5",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text"],
cost: {
input: 0.3,
output: 1.2,
cacheRead: 0.06,
cacheWrite: 0,
},
contextWindow: 204800,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"MiniMaxAI/MiniMax-M2.7": {
id: "MiniMaxAI/MiniMax-M2.7",
name: "MiniMax-M2.7",
@@ -42,28 +23,28 @@ export const TOGETHER_MODELS = {
contextWindow: 202752,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"Qwen/Qwen3-235B-A22B-Instruct-2507-tput": {
id: "Qwen/Qwen3-235B-A22B-Instruct-2507-tput",
name: "Qwen3 235B A22B Instruct 2507 FP8",
"MiniMaxAI/MiniMax-M3": {
id: "MiniMaxAI/MiniMax-M3",
name: "MiniMax-M3",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text"],
input: ["text", "image"],
cost: {
input: 0.2,
output: 0.6,
cacheRead: 0,
input: 0.3,
output: 1.2,
cacheRead: 0.06,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
contextWindow: 524288,
maxTokens: 250000,
} satisfies Model<"openai-completions">,
"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
id: "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8",
name: "Qwen3 Coder 480B A35B Instruct",
"Qwen/Qwen2.5-7B-Instruct-Turbo": {
id: "Qwen/Qwen2.5-7B-Instruct-Turbo",
name: "Qwen 2.5 7B Instruct Turbo",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
@@ -71,27 +52,26 @@ export const TOGETHER_MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 2,
output: 2,
input: 0.3,
output: 0.3,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
contextWindow: 32768,
maxTokens: 32768,
} satisfies Model<"openai-completions">,
"Qwen/Qwen3-Coder-Next-FP8": {
id: "Qwen/Qwen3-Coder-Next-FP8",
name: "Qwen3 Coder Next FP8",
"Qwen/Qwen3-235B-A22B-Instruct-2507-tput": {
id: "Qwen/Qwen3-235B-A22B-Instruct-2507-tput",
name: "Qwen3 235B A22B Instruct 2507 FP8",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false},
reasoning: false,
input: ["text"],
cost: {
input: 0.5,
output: 1.2,
input: 0.2,
output: 0.6,
cacheRead: 0,
cacheWrite: 0,
},
@@ -117,6 +97,25 @@ export const TOGETHER_MODELS = {
contextWindow: 262144,
maxTokens: 130000,
} satisfies Model<"openai-completions">,
"Qwen/Qwen3.5-9B": {
id: "Qwen/Qwen3.5-9B",
name: "Qwen3.5 9B",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text", "image"],
cost: {
input: 0.17,
output: 0.25,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 65536,
} satisfies Model<"openai-completions">,
"Qwen/Qwen3.6-Plus": {
id: "Qwen/Qwen3.6-Plus",
name: "Qwen3.6 Plus",
@@ -142,57 +141,18 @@ export const TOGETHER_MODELS = {
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false},
reasoning: false,
input: ["text"],
cost: {
input: 2.5,
output: 7.5,
input: 1.25,
output: 3.75,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 500000,
} satisfies Model<"openai-completions">,
"deepseek-ai/DeepSeek-V3": {
id: "deepseek-ai/DeepSeek-V3",
name: "DeepSeek-V3",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text"],
cost: {
input: 1.25,
output: 1.25,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"deepseek-ai/DeepSeek-V3-1": {
id: "deepseek-ai/DeepSeek-V3-1",
name: "DeepSeek V3.1",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text"],
cost: {
input: 0.6,
output: 1.7,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"deepseek-ai/DeepSeek-V4-Pro": {
id: "deepseek-ai/DeepSeek-V4-Pro",
name: "DeepSeek V4 Pro",
@@ -204,8 +164,8 @@ export const TOGETHER_MODELS = {
thinkingLevelMap: {"minimal":null,"low":null,"medium":null,"high":"high","xhigh":null},
input: ["text"],
cost: {
input: 2.1,
output: 4.4,
input: 1.74,
output: 3.48,
cacheRead: 0.2,
cacheWrite: 0,
},
@@ -241,8 +201,8 @@ export const TOGETHER_MODELS = {
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text", "image"],
cost: {
input: 0.2,
output: 0.5,
input: 0.39,
output: 0.97,
cacheRead: 0,
cacheWrite: 0,
},
@@ -267,25 +227,6 @@ export const TOGETHER_MODELS = {
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"moonshotai/Kimi-K2.5": {
id: "moonshotai/Kimi-K2.5",
name: "Kimi K2.5",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text", "image"],
cost: {
input: 0.5,
output: 2.8,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 262144,
} satisfies Model<"openai-completions">,
"moonshotai/Kimi-K2.6": {
id: "moonshotai/Kimi-K2.6",
name: "Kimi K2.6",
@@ -305,6 +246,25 @@ export const TOGETHER_MODELS = {
contextWindow: 262144,
maxTokens: 131000,
} satisfies Model<"openai-completions">,
"moonshotai/Kimi-K2.7-Code": {
id: "moonshotai/Kimi-K2.7-Code",
name: "Kimi K2.7 Code",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text"],
cost: {
input: 0.95,
output: 4,
cacheRead: 0.19,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"nvidia/nemotron-3-ultra-550b-a55b": {
id: "nvidia/nemotron-3-ultra-550b-a55b",
name: "Nemotron 3 Ultra 550B A55B",
@@ -343,6 +303,44 @@ export const TOGETHER_MODELS = {
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"openai/gpt-oss-20b": {
id: "openai/gpt-oss-20b",
name: "GPT OSS 20B",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":true,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"openai"},
reasoning: true,
thinkingLevelMap: {"off":null,"minimal":null},
input: ["text"],
cost: {
input: 0.05,
output: 0.2,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 131072,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"zai-org/GLM-5": {
id: "zai-org/GLM-5",
name: "GLM-5",
api: "openai-completions",
provider: "together",
baseUrl: "https://api.together.ai/v1",
compat: {"supportsStore":false,"supportsDeveloperRole":false,"supportsReasoningEffort":false,"maxTokensField":"max_tokens","supportsStrictMode":false,"supportsLongCacheRetention":false,"thinkingFormat":"together"},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":null,"medium":null},
input: ["text"],
cost: {
input: 1,
output: 3.2,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 202752,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"zai-org/GLM-5.1": {
id: "zai-org/GLM-5.1",
name: "GLM-5.1",
@@ -98,7 +98,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.39999999999999997,
input: 0.4,
output: 4,
cacheRead: 0,
cacheWrite: 0,
@@ -168,7 +168,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1,
output: 5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -268,7 +268,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.39999999999999997,
input: 0.4,
output: 4,
cacheRead: 0,
cacheWrite: 0,
@@ -285,8 +285,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.09999999999999999,
output: 0.39999999999999997,
input: 0.1,
output: 0.4,
cacheRead: 0.001,
cacheWrite: 0.125,
},
@@ -302,7 +302,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.39999999999999997,
input: 0.4,
output: 2.4,
cacheRead: 0.04,
cacheWrite: 0.5,
@@ -320,7 +320,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text", "image"],
cost: {
input: 0.6,
output: 3.5999999999999996,
output: 3.6,
cacheRead: 0,
cacheWrite: 0,
},
@@ -338,7 +338,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.5,
output: 3,
cacheRead: 0.09999999999999999,
cacheRead: 0.1,
cacheWrite: 0.625,
},
contextWindow: 1000000,
@@ -370,8 +370,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.39999999999999997,
output: 1.5999999999999999,
input: 0.4,
output: 1.6,
cacheRead: 0.08,
cacheWrite: 0.5,
},
@@ -404,7 +404,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.7999999999999999,
input: 0.8,
output: 4,
cacheRead: 0.08,
cacheWrite: 1,
@@ -412,25 +412,6 @@ export const VERCEL_AI_GATEWAY_MODELS = {
contextWindow: 200000,
maxTokens: 8192,
} satisfies Model<"anthropic-messages">,
"anthropic/claude-fable-5": {
id: "anthropic/claude-fable-5",
name: "Claude Fable 5",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
compat: {"forceAdaptiveThinking":true},
reasoning: true,
thinkingLevelMap: {"off":null,"xhigh":"xhigh"},
input: ["text", "image"],
cost: {
input: 10,
output: 50,
cacheRead: 1,
cacheWrite: 12.5,
},
contextWindow: 1000000,
maxTokens: 128000,
} satisfies Model<"anthropic-messages">,
"anthropic/claude-haiku-4.5": {
id: "anthropic/claude-haiku-4.5",
name: "Claude Haiku 4.5",
@@ -442,7 +423,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1,
output: 5,
cacheRead: 0.09999999999999999,
cacheRead: 0.1,
cacheWrite: 1.25,
},
contextWindow: 200000,
@@ -635,7 +616,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text"],
cost: {
input: 0.25,
output: 0.8999999999999999,
output: 0.9,
cacheRead: 0,
cacheWrite: 0,
},
@@ -653,7 +634,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.25,
output: 2,
cacheRead: 0.049999999999999996,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 256000,
@@ -838,8 +819,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.09999999999999999,
output: 0.39999999999999997,
input: 0.1,
output: 0.4,
cacheRead: 0.01,
cacheWrite: 0,
},
@@ -874,7 +855,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.5,
output: 3,
cacheRead: 0.049999999999999996,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -891,7 +872,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 2,
output: 12,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -942,7 +923,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 2,
output: 12,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -992,7 +973,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text", "image"],
cost: {
input: 0.14,
output: 0.39999999999999997,
output: 0.4,
cacheRead: 0,
cacheWrite: 0,
},
@@ -1010,7 +991,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.25,
output: 0.75,
cacheRead: 0.024999999999999998,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -1162,7 +1143,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text", "image"],
cost: {
input: 0.24,
output: 0.9700000000000001,
output: 0.97,
cacheRead: 0,
cacheWrite: 0,
},
@@ -1178,7 +1159,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.16999999999999998,
input: 0.17,
output: 0.66,
cacheRead: 0,
cacheWrite: 0,
@@ -1332,7 +1313,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text"],
cost: {
input: 0.3,
output: 0.8999999999999999,
output: 0.9,
cacheRead: 0,
cacheWrite: 0,
},
@@ -1348,7 +1329,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 0.39999999999999997,
input: 0.4,
output: 2,
cacheRead: 0,
cacheWrite: 0,
@@ -1365,7 +1346,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 0.09999999999999999,
input: 0.1,
output: 0.3,
cacheRead: 0,
cacheWrite: 0,
@@ -1382,7 +1363,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 0.09999999999999999,
input: 0.1,
output: 0.3,
cacheRead: 0,
cacheWrite: 0,
@@ -1399,8 +1380,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 0.09999999999999999,
output: 0.09999999999999999,
input: 0.1,
output: 0.1,
cacheRead: 0,
cacheWrite: 0,
},
@@ -1433,7 +1414,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.39999999999999997,
input: 0.4,
output: 2,
cacheRead: 0,
cacheWrite: 0,
@@ -1467,13 +1448,13 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 0.02,
output: 0.04,
input: 0.15,
output: 0.15,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 131072,
maxTokens: 131072,
contextWindow: 128000,
maxTokens: 128000,
} satisfies Model<"anthropic-messages">,
"mistral/mistral-small": {
id: "mistral/mistral-small",
@@ -1484,7 +1465,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.09999999999999999,
input: 0.1,
output: 0.3,
cacheRead: 0,
cacheWrite: 0,
@@ -1535,7 +1516,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text"],
cost: {
input: 0.5700000000000001,
input: 0.57,
output: 2.3,
cacheRead: 0,
cacheWrite: 0,
@@ -1560,40 +1541,6 @@ export const VERCEL_AI_GATEWAY_MODELS = {
contextWindow: 262114,
maxTokens: 262114,
} satisfies Model<"anthropic-messages">,
"moonshotai/kimi-k2-thinking-turbo": {
id: "moonshotai/kimi-k2-thinking-turbo",
name: "Kimi K2 Thinking Turbo",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: true,
input: ["text"],
cost: {
input: 1.15,
output: 8,
cacheRead: 0.15,
cacheWrite: 0,
},
contextWindow: 262114,
maxTokens: 262114,
} satisfies Model<"anthropic-messages">,
"moonshotai/kimi-k2-turbo": {
id: "moonshotai/kimi-k2-turbo",
name: "Kimi K2 Turbo",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: false,
input: ["text"],
cost: {
input: 1.15,
output: 8,
cacheRead: 0.15,
cacheWrite: 0,
},
contextWindow: 256000,
maxTokens: 16384,
} satisfies Model<"anthropic-messages">,
"moonshotai/kimi-k2.5": {
id: "moonshotai/kimi-k2.5",
name: "Kimi K2.5",
@@ -1605,7 +1552,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.6,
output: 3,
cacheRead: 0.09999999999999999,
cacheRead: 0.1,
cacheWrite: 0,
},
contextWindow: 262114,
@@ -1628,6 +1575,40 @@ export const VERCEL_AI_GATEWAY_MODELS = {
contextWindow: 262000,
maxTokens: 262000,
} satisfies Model<"anthropic-messages">,
"moonshotai/kimi-k2.7-code": {
id: "moonshotai/kimi-k2.7-code",
name: "Kimi K2.7 Code",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.95,
output: 4,
cacheRead: 0.19,
cacheWrite: 0,
},
contextWindow: 256000,
maxTokens: 32768,
} satisfies Model<"anthropic-messages">,
"moonshotai/kimi-k2.7-code-highspeed": {
id: "moonshotai/kimi-k2.7-code-highspeed",
name: "Kimi K2.7 Code High Speed",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: true,
input: ["text", "image"],
cost: {
input: 1.9,
output: 8,
cacheRead: 0.38,
cacheWrite: 0,
},
contextWindow: 262144,
maxTokens: 32768,
} satisfies Model<"anthropic-messages">,
"nvidia/nemotron-3-super-120b-a12b": {
id: "nvidia/nemotron-3-super-120b-a12b",
name: "NVIDIA Nemotron 3 Super 120B A12B",
@@ -1671,7 +1652,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.19999999999999998,
input: 0.2,
output: 0.6,
cacheRead: 0,
cacheWrite: 0,
@@ -1689,7 +1670,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text"],
cost: {
input: 0.06,
output: 0.22999999999999998,
output: 0.23,
cacheRead: 0,
cacheWrite: 0,
},
@@ -1739,9 +1720,9 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.39999999999999997,
output: 1.5999999999999999,
cacheRead: 0.09999999999999999,
input: 0.4,
output: 1.6,
cacheRead: 0.1,
cacheWrite: 0,
},
contextWindow: 1047576,
@@ -1756,9 +1737,9 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.09999999999999999,
output: 0.39999999999999997,
cacheRead: 0.024999999999999998,
input: 0.1,
output: 0.4,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 1047576,
@@ -1860,7 +1841,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.25,
output: 2,
cacheRead: 0.024999999999999998,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 400000,
@@ -1875,8 +1856,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.049999999999999996,
output: 0.39999999999999997,
input: 0.05,
output: 0.4,
cacheRead: 0.005,
cacheWrite: 0,
},
@@ -1945,7 +1926,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.25,
output: 2,
cacheRead: 0.024999999999999998,
cacheRead: 0.025,
cacheWrite: 0,
},
contextWindow: 400000,
@@ -2139,7 +2120,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
thinkingLevelMap: {"xhigh":"xhigh"},
input: ["text", "image"],
cost: {
input: 0.19999999999999998,
input: 0.2,
output: 1.25,
cacheRead: 0.02,
cacheWrite: 0,
@@ -2227,8 +2208,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text"],
cost: {
input: 0.049999999999999996,
output: 0.19999999999999998,
input: 0.05,
output: 0.2,
cacheRead: 0,
cacheWrite: 0,
},
@@ -2388,6 +2369,23 @@ export const VERCEL_AI_GATEWAY_MODELS = {
contextWindow: 200000,
maxTokens: 8000,
} satisfies Model<"anthropic-messages">,
"sakana/fugu-ultra": {
id: "sakana/fugu-ultra",
name: "Fugu Ultra",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: true,
input: ["text", "image"],
cost: {
input: 5,
output: 30,
cacheRead: 0.5,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 1000000,
} satisfies Model<"anthropic-messages">,
"stepfun/step-3.5-flash": {
id: "stepfun/step-3.5-flash",
name: "StepFun 3.5 Flash",
@@ -2399,8 +2397,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 0.09,
output: 0.3,
cacheRead: 0,
cacheWrite: 0.02,
cacheRead: 0.02,
cacheWrite: 0,
},
contextWindow: 262114,
maxTokens: 262114,
@@ -2414,7 +2412,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.19999999999999998,
input: 0.2,
output: 1.15,
cacheRead: 0.04,
cacheWrite: 0,
@@ -2431,9 +2429,9 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: false,
input: ["text", "image"],
cost: {
input: 0.19999999999999998,
input: 0.2,
output: 0.5,
cacheRead: 0.049999999999999996,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -2448,9 +2446,9 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text", "image"],
cost: {
input: 0.19999999999999998,
input: 0.2,
output: 0.5,
cacheRead: 0.049999999999999996,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -2467,7 +2465,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1.25,
output: 2.5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 2000000,
@@ -2484,7 +2482,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1.25,
output: 2.5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 2000000,
@@ -2501,7 +2499,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1.25,
output: 2.5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 2000000,
@@ -2518,7 +2516,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1.25,
output: 2.5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 2000000,
@@ -2535,7 +2533,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1.25,
output: 2.5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 2000000,
@@ -2552,7 +2550,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1.25,
output: 2.5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 2000000,
@@ -2569,7 +2567,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1.25,
output: 2.5,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -2586,7 +2584,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1,
output: 2,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 256000,
@@ -2601,7 +2599,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text"],
cost: {
input: 0.09999999999999999,
input: 0.1,
output: 0.3,
cacheRead: 0.01,
cacheWrite: 0,
@@ -2620,7 +2618,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
cost: {
input: 1,
output: 3,
cacheRead: 0.19999999999999998,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 1000000,
@@ -2686,7 +2684,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
reasoning: true,
input: ["text"],
cost: {
input: 0.19999999999999998,
input: 0.2,
output: 1.1,
cacheRead: 0.03,
cacheWrite: 0,
@@ -2704,7 +2702,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text", "image"],
cost: {
input: 0.6,
output: 1.7999999999999998,
output: 1.8,
cacheRead: 0.11,
cacheWrite: 0,
},
@@ -2738,8 +2736,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text", "image"],
cost: {
input: 0.3,
output: 0.8999999999999999,
cacheRead: 0.049999999999999996,
output: 0.9,
cacheRead: 0.05,
cacheWrite: 0,
},
contextWindow: 128000,
@@ -2789,7 +2787,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text"],
cost: {
input: 0.07,
output: 0.39999999999999997,
output: 0.4,
cacheRead: 0,
cacheWrite: 0,
},
@@ -2806,7 +2804,7 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text"],
cost: {
input: 0.06,
output: 0.39999999999999997,
output: 0.4,
cacheRead: 0.01,
cacheWrite: 0,
},
@@ -2823,8 +2821,8 @@ export const VERCEL_AI_GATEWAY_MODELS = {
input: ["text"],
cost: {
input: 1,
output: 3.1999999999999997,
cacheRead: 0.19999999999999998,
output: 3.2,
cacheRead: 0.2,
cacheWrite: 0,
},
contextWindow: 202800,
@@ -2864,6 +2862,23 @@ export const VERCEL_AI_GATEWAY_MODELS = {
contextWindow: 202800,
maxTokens: 64000,
} satisfies Model<"anthropic-messages">,
"zai/glm-5.2": {
id: "zai/glm-5.2",
name: "GLM 5.2",
api: "anthropic-messages",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh",
reasoning: true,
input: ["text"],
cost: {
input: 1.5,
output: 4.5,
cacheRead: 0.3,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 128000,
} satisfies Model<"anthropic-messages">,
"zai/glm-5v-turbo": {
id: "zai/glm-5v-turbo",
name: "GLM 5V Turbo",
@@ -76,6 +76,25 @@ export const ZAI_CODING_CN_MODELS = {
contextWindow: 200000,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"glm-5.2": {
id: "glm-5.2",
name: "GLM-5.2",
api: "openai-completions",
provider: "zai-coding-cn",
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
compat: {"supportsDeveloperRole":false,"thinkingFormat":"zai","supportsReasoningEffort":true,"zaiToolStream":true},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":"high","medium":"high","high":"high","xhigh":"max"},
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"glm-5v-turbo": {
id: "glm-5v-turbo",
name: "GLM-5V-Turbo",
+19
View File
@@ -76,6 +76,25 @@ export const ZAI_MODELS = {
contextWindow: 200000,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"glm-5.2": {
id: "glm-5.2",
name: "GLM-5.2",
api: "openai-completions",
provider: "zai",
baseUrl: "https://api.z.ai/api/coding/paas/v4",
compat: {"supportsDeveloperRole":false,"thinkingFormat":"zai","supportsReasoningEffort":true,"zaiToolStream":true},
reasoning: true,
thinkingLevelMap: {"minimal":null,"low":"high","medium":"high","high":"high","xhigh":"max"},
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
contextWindow: 1000000,
maxTokens: 131072,
} satisfies Model<"openai-completions">,
"glm-5v-turbo": {
id: "glm-5v-turbo",
name: "GLM-5V-Turbo",