merge: sync upstream v0.5.81 into MIBP fork

# Conflicts:
#	.gitignore
#	Dockerfile
#	open-sse/handlers/chatCore.js
#	open-sse/providers/registry/cline.js
#	open-sse/providers/registry/index.js
#	open-sse/services/usage.js
#	open-sse/utils/streamHandler.js
#	package.json
#	src/app/(dashboard)/dashboard/profile/page.js
#	src/app/(dashboard)/dashboard/providers/[id]/page.js
This commit is contained in:
MUH. IQRAM BAHRING
2026-09-19 11:35:34 +08:00
208 changed files with 13062 additions and 977 deletions
+169 -25
View File
@@ -116,6 +116,16 @@ export const MODEL_CAPABILITIES = {
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
// DeepSeek V4.1-Flash is natively multimodal — models.dev lists
// opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream
// the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the
// same image capability as the exp id above. "deepseek-flash" is the GA id on the
// DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose
// 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry
// short-circuits the pattern table, so a vision-only delta would drop them.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
"deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 },
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
"coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
@@ -131,6 +141,8 @@ export const MODEL_CAPABILITIES = {
// via OpenAI Responses input_image; reasoning supports up to xhigh.
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
// OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output
"union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 },
};
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
@@ -154,6 +166,7 @@ export const PROVIDER_CAPABILITIES = {
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
},
"codex": {
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
@@ -178,39 +191,95 @@ export const PROVIDER_CAPABILITIES = {
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
// (see registry thinkingFormat). `onlyReasoning` models can't turn thinking
// off → thinkingCanDisable:false (clamped to minimal instead of disabled).
// (see registry thinkingFormat). For thinkingCanDisable use the server's
// reasoning.canDisableThinking flag — see the note in the codebuddy-cn block
// below; it is NOT the inverse of onlyReasoning.
"codebuddy-cn": {
"glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 },
// maxOutput 64000 per both the plugin-baked fallback and the live server
// table (the old 38000 had no source and truncated output).
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 },
"glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 },
"minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 },
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
// hy3/hy3-x: 256K official (192K conservative, matches hy3-preview); hy4-preview: 1M official.
// glm-5.3: 1M (GLM-5.x gen); glm-5.3-flash window unverified (200K conservative).
// Per-model values mirror the server's product-config payload (the plugin
// fetches it from copilot.tencent.com; the `models[]` entries carry
// maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
// maxInputTokens, maxOutput = maxOutputTokens. Where the server and the
// plugin-baked fallback disagree, the server table wins.
// ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking —
// it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is
// on by default"; canDisableThinking means "it CAN be turned off". glm-5.3
// and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so
// their thinking is switchable; the hy* models are forced always-on.
"hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"hy3-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
"hy4-preview-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
"glm-5.3": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.3-flash": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 },
"glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
// deepseek-v4.1-flash replaces v4-flash (dropped from the server list;
// the old endpoint still answers 200 but the published list is the
// contract). maxOutput 128000 per the server's product-config payload.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors
// the codebuddy-cn entry (the openai-style reasoning_effort format matters:
// the generic *deepseek-v4* pattern would otherwise pick the vendor-native
// "deepseek" thinking shape, which the CodeBuddy gateway does not accept).
"codebuddy-intl": {
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
// registry `name` is display-only and capability lookup matches on the raw
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
// (200K) without this map. contextWindow follows the real model family's
// spec: the /algo/api/v2/model/list max_input_tokens under-reports some
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
// max_output_tokens arrives as 0 for every model, so outputs are
// best-guess from the real model family. Vision tags below follow the
// upstream is_vl flag. The executor uploads inlined images to
// /api/v2/image/upload and leaves image_urls/chat_context.imageUrls null
// (same as qodercli). reasoning:true on all of them — every model can
// reason; the upstream is_reasoning flag only drives model_config selection.
// thinkingFormat keeps the true-model family for documentation/UI, but
// thinkingCanDisable:false everywhere: the executor only forwards
// messages/tools/max_tokens, and thinking is fixed upstream via
// modelConfig.is_reasoning — client thinking intent is dropped, so "none"
// must never be offered as an option.
"qoder": {
"ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5
"performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5
"dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro
"dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash
"gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3
"gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash
"kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3
"kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code
"mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3
"qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max
"qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus
"qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash
"qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max
},
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
"poolside": {
"laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 },
"laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 },
},
// Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge
// the library page publishes for this model (text+image in, 1M context).
// ponytail: thinkingFormat stays "deepseek" to preserve today's body shape;
// Ollama's native toggle is the top-level `think` field (bool or
// low/medium/high/max), which no format in thinkingUnified.js emits yet —
// openai-to-ollama.js drops it. Wire a "think" format when thinking on
// Ollama Cloud is actually needed.
"ollama": {
"deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
},
};
/**
@@ -247,6 +316,9 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } },
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
@@ -312,7 +384,11 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
// ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } },
// v4.1+ has real image input (probed live on Alibaba MaaS: correct color
// read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore
// them (answered "Unknown"), so vision stays scoped to v4.* dotted releases.
{ pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } },
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } },
{ pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
{ pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
{ pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } },
@@ -377,14 +453,27 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
// Catalog lookups, installed by the server at startup. Left as no-ops in the
// browser bundle, where there is no file to read.
//
// The server bundles this module into every route chunk that needs it, and each
// copy carries its own module state, so an install landing in the copy the
// startup hook imported stays invisible to the copy resolving requests. The slot
// lives on globalThis instead; the local binding is the fast path.
let catalogSource = null;
/**
* Install the synced catalog reader (server only).
* @param {{ getModalities: Function, getLimits: Function } | null} source
* @param {{ getModalities: (provider: string, model: string) => object|null,
* getLimits: (provider: string, model: string) => object|null } | null} source
*/
export function setCatalogSource(source) {
catalogSource = source;
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source;
}
function getCatalogSource() {
if (catalogSource) return catalogSource;
if (typeof globalThis === "undefined") return null;
return (catalogSource = globalThis.__9rCatalogSource || null);
}
// Apply the synced catalog + name heuristic on top of a table-resolved result.
@@ -393,15 +482,16 @@ export function setCatalogSource(source) {
function refine(base, provider, model) {
const result = { ...DEFAULT_CAPABILITIES, ...base };
if (catalogSource) {
const modalities = catalogSource.getModalities(model);
const source = getCatalogSource();
if (source) {
const modalities = source.getModalities(provider, model);
if (modalities) {
for (const key of MODALITY_KEYS) {
if (modalities[key] === true) result[key] = true;
}
}
const limits = catalogSource.getLimits(provider, model);
const limits = source.getLimits(provider, model);
if (limits) {
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
@@ -413,12 +503,66 @@ function refine(base, provider, model) {
return result;
}
// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models
// default to vision; only this denylist stays text-only.
const COMMANDCODE_TEXT_ONLY = new Set([
"deepseek/deepseek-v4-pro",
"deepseek/deepseek-v4-flash",
"deepseek/deepseek-v4-flash-fast",
"zai-org/glm-5.3",
"zai-org/glm-5.2",
"zai-org/glm-5.2-fast",
"zai-org/glm-5.1",
"zai-org/glm-5",
"minimaxai/minimax-m2.7",
"minimax/minimax-m2.7-free",
"minimaxai/minimax-m2.5",
"xiaomi/mimo-v2.5-pro",
"qwen/qwen3.6-max-preview",
"qwen/qwen3.7-max",
"meituan/longcat-2.0:free",
"stepfun/step-3.5-flash",
"tencent/hy4-preview",
"tencent/hy3",
"tencent/hy3-paid",
"nvidia/nemotron-3-ultra-550b-a55b",
"poolside/laguna-s-2.1-free",
"inclusionai/ling-3.0-flash-free",
"inclusionai/ling-3.0-flash-sante:free",
]);
function isCommandCodeTextOnly(model) {
const key = String(model || "").toLowerCase();
if (COMMANDCODE_TEXT_ONLY.has(key)) return true;
for (const id of COMMANDCODE_TEXT_ONLY) {
const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
if (key === base || key.endsWith("/" + base)) return true;
}
return false;
}
export function getCapabilitiesForModel(provider, model) {
if (!model) return { ...DEFAULT_CAPABILITIES };
// Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7".
const baseModel = model.includes("/") ? model.split("/").pop() : model;
// CommandCode wire is /alpha/generate for every model. Family patterns
// (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here.
if (provider === "commandcode" || provider === "cmc") {
const providerCaps = PROVIDER_CAPABILITIES.commandcode;
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
return {
...DEFAULT_CAPABILITIES,
reasoning: true,
thinkingFormat: "commandcode",
thinkingEffortSupported: true,
vision: !isCommandCodeTextOnly(model),
contextWindow: 1000000,
maxOutput: 384000,
};
}
// 1. Provider-specific override
if (provider) {
const providerCaps = PROVIDER_CAPABILITIES[provider];
+16 -6
View File
@@ -13,6 +13,11 @@ export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
// Trimmed upstream catalog, read by the add-models skill (not by the router).
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
// Schema of the file this module reads. The writer stamps it; a file carrying an
// older value predates provider-scoped modality keys, and its flat keys are not
// looked up here, so the sync rebuilds it instead of asking upstream for a 304.
export const CATALOG_VERSION = 2;
const EMPTY = { models: {}, providers: {} };
let cache = EMPTY;
let cachedMtime = -1;
@@ -45,14 +50,19 @@ function load() {
return cache;
}
// Modality is a property of the model itself — any gateway serving it inherits
// the same image/video/pdf support, so this is keyed by model id alone.
export function getCatalogModalities(model) {
return load().models[baseId(model)] || null;
// Modalities are recorded per gateway upstream, and gateways disagree about the
// same weights — some do not proxy images at all — so the key is provider +
// model, in the local provider id space, exactly like the limits below. Keying
// by model id alone made short ids collide across vendors: "auto", "free" and
// "efficient" are router modes in one catalog and model names in another, and a
// request to the router mode inherited a stranger's vision.
export function getCatalogModalities(provider, model) {
if (!provider) return null;
return load().models[`${provider}:${baseId(model)}`] || null;
}
// Context and output limits are a property of the gateway, not the model: each
// one truncates differently, so these stay keyed by provider + model.
// Context and output limits are a property of the gateway too: each one
// truncates differently, so these stay keyed by provider + model.
export function getCatalogLimits(provider, model) {
const byProvider = provider && load().providers[provider];
if (!byProvider) return null;
+3
View File
@@ -53,6 +53,7 @@ export const MODEL_PRICING = {
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
@@ -110,6 +111,8 @@ export const MODEL_PRICING = {
"deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 },
// === GLM ===
@@ -37,6 +37,7 @@ export default {
},
usage: {
quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
quotaSummaryApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:retrieveUserQuotaSummary`,
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
tokenUrl: "https://oauth2.googleapis.com/token",
},
+6 -3
View File
@@ -20,6 +20,8 @@ export default {
authModes: [
"apikey",
],
passthroughModels: true,
modelsFetcher: { url: "https://api.airforce/v1/models", type: "airforce-free" },
transport: {
baseUrl: "https://api.airforce/v1/chat/completions",
validateUrl: "https://api.airforce/v1/models",
@@ -27,10 +29,11 @@ export default {
"HTTP-Referer": "https://endpoint-proxy.local",
"X-Title": "Endpoint Proxy",
},
forceStream: true,
},
models: [
{ id: "anthropic/claude-3.7-sonnet", name: "Claude 3.7 Sonnet (Free)", contextLength: 200000 },
{ id: "moonshot/kimi-k2.6", name: "Kimi K2.6 (Free)", contextLength: 262144 },
{ id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash (Free)", contextLength: 1048576 },
{ id: "gpt-oss-120b", name: "GPT-OSS 120B (Free)", contextLength: 131072 },
{ id: "gpt-oss-20b", name: "GPT-OSS 20B (Free)", contextLength: 131072 },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code (Free)", contextLength: 262144 },
],
};
+2
View File
@@ -23,6 +23,8 @@ export default {
"HTTP-Referer": "https://cline.bot",
"X-Title": "Cline",
},
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
quirks: { clineEnvelope: true },
tokenUrl: "https://api.cline.bot/api/v1/auth/token",
refreshUrl: "https://api.cline.bot/api/v1/auth/refresh",
auth: {
+6 -1
View File
@@ -14,7 +14,10 @@ export default {
},
},
category: "oauth",
authModes: ["oauth", "apikey"],
// ClinePass authenticates with a plain API key from app.cline.bot/settings/api-keys
// (category "apikey"). The OAuth extension flow used by Cline does not issue
// tokens that the ClinePass API consumer endpoint accepts (HTTP 401) — see #2333.
authModes: ["apikey", "oauth"],
hasOAuth: true,
transport: {
baseUrl: "https://api.cline.bot/api/v1/chat/completions",
@@ -22,6 +25,8 @@ export default {
"HTTP-Referer": "https://cline.bot",
"X-Title": "Cline",
},
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
quirks: { clineEnvelope: true },
auth: {
combined: true,
header: "Authorization",
+12 -11
View File
@@ -47,27 +47,28 @@ export default {
models: [
{ id: "glm-5.2", name: "GLM-5.2" },
{ id: "glm-5.1", name: "GLM-5.1" },
{ id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" },
{ id: "glm-5v-turbo", name: "GLM-5v-Turbo" },
{ id: "minimax-m3", name: "MiniMax-M3" },
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
{ id: "kimi-k2.7", name: "Kimi-K2.7-Code" },
{ id: "kimi-k2.6", name: "Kimi-K2.6" },
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
// "-x" suffix = paid tier of the same model (free id rides the promo quota:
// hy3 free until 2026-08-31, hy4-preview until 2026-09-10). Server model table
// seen in client logs 2026-08-30; glm-5.0 / glm-4.7 removed (API 11102 dead).
{ id: "hy3-preview", name: "Hy3 Preview" },
// Catalog mirrors the server's product-config payload (the plugin fetches
// it from copilot.tencent.com). Models the server no longer publishes are
// removed even when the chat endpoint still answers them — the published
// list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x
// (endpoint returns 11102 "model service info not found"), plus
// glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview /
// deepseek-v3-2-volc (absent from the server list, though still answering
// 200) and hy3-x (paid tier, not used here). deepseek-v4-flash removed
// 2026-09: replaced server-side by deepseek-v4.1-flash (same low/high/
// xhigh efforts; endpoint still answers 200 but the list is the contract).
// "-x" suffix = paid tier of the same model (free id rides the promo quota).
{ id: "hy3", name: "Hy3" },
{ id: "hy3-x", name: "Hy3 (Paid)" },
{ id: "hy4-preview", name: "Hy4-Preview" },
{ id: "hy4-preview-x", name: "Hy4-Preview (Paid)" },
{ id: "glm-5.3", name: "GLM-5.3" },
{ id: "glm-5.3-flash", name: "GLM-5.3-Flash" },
{ id: "kimi-k3-1", name: "Kimi-K3" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
],
oauth: {
baseUrl: "https://copilot.tencent.com",
@@ -58,7 +58,9 @@ export default {
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
{ id: "hy3-preview", name: "Hy3 Preview" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
// deepseek-v4-flash replaced server-side by deepseek-v4.1-flash (same
// catalog as CN; the old endpoint still answers 200 but the list is the contract).
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
],
oauth: {
+18 -1
View File
@@ -1,5 +1,9 @@
import { withCodexReviewModels } from "../models/helpers.js";
// Codex CLI version seen by OpenAI's backend — single source for the Version /
// User-Agent identity headers. Bump when the installed codex CLI is upgraded.
const CODEX_CLI_VERSION = "0.154.0";
export default {
id: "codex",
priority: 30,
@@ -34,9 +38,10 @@ export default {
baseUrl: "https://chatgpt.com/backend-api/codex/responses",
format: "openai-responses",
forceStream: true,
cliVersion: CODEX_CLI_VERSION,
headers: {
originator: "codex_cli_rs",
"User-Agent": "codex_cli_rs/0.136.0",
"User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`,
},
usage: {
url: "https://chatgpt.com/backend-api/wham/usage",
@@ -45,6 +50,7 @@ export default {
},
},
models: [
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
@@ -59,6 +65,17 @@ export default {
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
// Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived
// from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398).
{ id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" },
{ id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2", name: "GPT Image 2", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-1.5", name: "GPT Image 1.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
@@ -40,4 +40,8 @@ export default {
{ id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" },
{ id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" },
],
features: {
usage: true,
usageApikey: true,
},
};
+16
View File
@@ -25,6 +25,21 @@ export default {
reasoningInject: {
scope: "all",
},
quirks: {
// DeepSeek's Anthropic-compatible endpoint
// (https://api.deepseek.com/anthropic/v1/messages) accepts ONLY the
// built-in web_search_* tools and rejects client-defined `custom` tools
// (MCP / Read / Bash / etc.) with HTTP 400
// "tools[0]: unknown variant `custom`, expected
// `web_search_20250305` or `web_search_20260209`".
//
// Declaring this whitelist makes prepareClaudeRequest() forward only
// web_search_* tools and strip everything else before sending, so MCP /
// function tools are dropped instead of failing the whole request.
// DeepSeek's OpenAI-compatible transport is unaffected (targetFormat
// there is "openai", not "claude", so prepareClaudeRequest is not run).
claudeSupportedToolTypes: ["web_search_20250305", "web_search_20260209"],
},
},
// Multi-endpoint: pick the transport matching client sourceFormat to skip translation.
transports: [
@@ -44,6 +59,7 @@ export default {
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash" },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
+1
View File
@@ -25,6 +25,7 @@ export default {
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM-4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
+1
View File
@@ -49,6 +49,7 @@ export default {
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM 4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
@@ -22,6 +22,7 @@ export default {
headers: { ...CLAUDE_API_HEADERS },
quirks: {
dropOutputConfig: true,
requireClaudeToolType: true,
},
reasoningInject: {
scope: "all",
+1
View File
@@ -22,6 +22,7 @@ export default {
headers: { ...CLAUDE_API_HEADERS },
quirks: {
dropOutputConfig: true,
requireClaudeToolType: true,
},
reasoningInject: {
scope: "all",
+1
View File
@@ -30,6 +30,7 @@ export default {
{ id: "glm-4.7-flash", name: "GLM 4.7 Flash" },
{ id: "qwen3.5", name: "Qwen3.5" },
{ id: "minimax-m3", name: "MiniMax M3" },
{ id: "deepseek-v4.1-flash:cloud", name: "DeepSeek V4.1 Flash" },
],
serviceKinds: ["llm", "webFetch"],
fetchConfig: {
+3
View File
@@ -57,6 +57,9 @@ export default {
{ id: "whisper-1", name: "Whisper 1", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-4o-transcribe", name: "GPT-4o Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-4o-mini-transcribe", name: "GPT-4o Mini Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-image-2.5", name: "GPT Image 2.5", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-1", name: "GPT Image 1", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "dall-e-3", name: "DALL-E 3", params: ["size","quality","style","response_format"], kind: "image" },
{ id: "dall-e-2", name: "DALL-E 2", params: ["n","size","response_format"], kind: "image" },
+23 -1
View File
@@ -13,7 +13,7 @@ export default {
textIcon: "OC",
website: "https://opencode.ai/auth",
notice: {
text: "OpenCode Go subscription: $5/mo (then 0/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
text: "OpenCode Go subscription: $5/mo (then 10/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
apiKeyUrl: "https://opencode.ai/auth",
},
},
@@ -21,6 +21,9 @@ export default {
transport: {
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
headers: {},
usage: {
url: "https://opencode.ai/zen/go/v1/usage",
},
},
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
// translation. Guarded per-model by `supportedFormats` (see chatCore) because
@@ -30,22 +33,41 @@ export default {
{ format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } },
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
],
// supportedFormats follow the endpoint table in https://opencode.ai/docs/go/
models: [
{ id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
{ id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] },
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
{ id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] },
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] },
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
{ id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] },
{ id: "hy3", name: "Hy3", supportedFormats: ["openai"] },
// Served by /zen/go/v1/responses only — the responses-only entry forces chatCore
// past the sourceFormat-matched transports into translation (see chatCore guard).
{ id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
],
features: {
usage: true,
usageApikey: true,
},
};
+6 -2
View File
@@ -17,13 +17,17 @@ export default {
headers: {
"x-opencode-client": "desktop",
},
forceStream: true,
noAuth: true,
quirks: {
forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"],
},
},
models: [
// Muse Spark models are served by /zen/v1/responses; the rest stay on
// /chat/completions, so the format is declared per-model, not per-provider.
// Endpoint formats differ per model, so declare non-chat models explicitly.
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
{ id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" },
],
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
passthroughModels: true,
+10 -1
View File
@@ -40,8 +40,11 @@ export default {
{ id: "openai/gpt-image-1", name: "GPT Image 1 (via OpenRouter)", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "google/imagen-3.0-generate-002", name: "Imagen 3 (via OpenRouter)", params: ["n","size"], kind: "image" },
{ id: "black-forest-labs/FLUX.1-schnell", name: "FLUX.1 Schnell (via OpenRouter)", params: ["n","size"], kind: "image" },
{ id: "google/veo-3.1", name: "Veo 3.1 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "openai/sora-2-pro", name: "Sora 2 Pro (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "bytedance/seedance-2.0", name: "Seedance 2.0 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
],
serviceKinds: ["llm","embedding","tts","imageToText"],
serviceKinds: ["llm","embedding","tts","imageToText","video"],
ttsConfig: {
baseUrl: "https://openrouter.ai/api/v1/chat/completions",
defaultModel: "openai/gpt-4o-mini-tts",
@@ -57,6 +60,12 @@ export default {
baseUrl: "https://openrouter.ai/api/v1/images/generations",
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
},
// Async video jobs (POST /videos → { id, status }, GET /videos/{id} polls).
// Docs: https://openrouter.ai/docs/api/api-reference/videos
videoConfig: {
baseUrl: "https://openrouter.ai/api/v1/videos",
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
},
modelsFetcher: { url: "https://openrouter.ai/api/v1/models", type: "openrouter-free" },
passthroughModels: true,
};
+5 -2
View File
@@ -30,12 +30,15 @@ export default {
{ id: "auto", name: "Auto" },
{ id: "performance", name: "Performance" },
{ id: "efficient", name: "Efficient" },
{ id: "qmodel_preview", name: "Qwen3.8-Max-Preview" },
{ id: "lite", name: "Lite" },
{ id: "qmodel_38max", name: "Qwen3.8-Max" },
{ id: "qmodel_latest", name: "Qwen3.7-Max" },
{ id: "qmodel", name: "Qwen3.7-Plus" },
{ id: "qfmodel", name: "Qwen3.8-Flash" },
{ id: "kmodel_latest", name: "Kimi-K3" },
{ id: "kmodel", name: "Kimi-K2.7-Code" },
{ id: "gm51model", name: "GLM-5.2" },
{ id: "gmodel", name: "GLM-5.3" },
{ id: "gfmodel", name: "GLM-5.3-Flash" },
{ id: "dmodel", name: "DeepSeek-V4-Pro" },
{ id: "dfmodel", name: "DeepSeek-V4-Flash" },
{ id: "mmodel", name: "MiniMax-M3" },
+8 -1
View File
@@ -27,6 +27,13 @@ export default {
{ id: "gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview" },
{ id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
{ id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" },
{ id: "veo-3.1-generate-preview", name: "Veo 3.1 (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
{ id: "veo-3.1-fast-generate-preview", name: "Veo 3.1 Fast (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
{ id: "veo-3.0-generate-001", name: "Veo 3", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
{ id: "veo-2.0-generate-001", name: "Veo 2", params: ["duration","aspect_ratio","negative_prompt","seed","storage_uri"], kind: "video" },
],
serviceKinds: ["llm","imageToText"],
serviceKinds: ["llm","imageToText","video"],
// Veo via predictLongRunning + fetchPredictOperation (adapter: handlers/videoProviders/vertex.js).
// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation
videoConfig: { baseUrl: "https://aiplatform.googleapis.com" },
};
+31 -1
View File
@@ -1,11 +1,19 @@
import { CLAUDE_API_HEADERS } from "../shared.js";
// Dual auth (same pattern as kimi):
// - API key (sk-...) → cloud API on api.xiaomimimo.com
// - Desktop account/OAuth → same cloud host, plus the Desktop-exclusive Preview
// models served by the account-service route on mimo-server-cn.xiaomimimo.com
// (authorized by a Xiaomi account session cookie, not the key).
// Endpoint is picked per model in the executor, same as opencode-go's /responses split.
export default {
id: "xiaomi-mimo",
priority: 290,
alias: "xiaomi-mimo",
aliases: [
"mimo",
"mimo-desktop",
"xmd",
],
uiAlias: "mimo",
display: {
@@ -16,9 +24,12 @@ export default {
website: "https://xiaomimimo.com",
notice: {
apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys",
signupUrl: "https://mimo.xiaomimimo.com/desktop/invite/",
},
},
category: "apikey",
category: "oauth",
authModes: ["oauth", "apikey"],
hasOAuth: true,
serviceKinds: ["llm", "tts"],
transport: {
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
@@ -39,6 +50,11 @@ export default {
},
],
models: [
// Desktop-exclusive — served by the account-service route, which only accepts
// OpenAI format, so supportedFormats pins them to the openai transport.
{ id: "mimo-x-pro-preview", name: "MiMo-X-Pro-Preview", upstreamModelId: "xiaomi/mimo-x-pro-preview", supportedFormats: ["openai"] },
{ id: "mimo-x-flash-preview", name: "MiMo-X-Flash-Preview", upstreamModelId: "xiaomi/mimo-x-flash-preview", supportedFormats: ["openai"] },
// Cloud API models (api.xiaomimimo.com/v1)
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
{ id: "mimo-v2.5", name: "MiMo V2.5" },
{ id: "mimo-v2-omni", name: "MiMo V2 Omni" },
@@ -51,4 +67,18 @@ export default {
authHeader: "bearer",
format: "xiaomi-mimo-tts",
},
features: {
usage: true,
usageApikey: true,
},
// Custom OAuth — non-standard ECDH encrypted-callback flow.
// Handled by the Xiaomi MiMo OAuth service, not the generic PKCE pipeline.
oauth: {
custom: true,
authorizeUrl: "https://platform.xiaomimimo.com/authorize",
// The callback carries ?u=<ECDH-encrypted payload> instead of ?code=.
// Decryption yields { uid, sk, url }.
callbackParam: "u",
kn: "mimocode",
},
};
+1 -2
View File
@@ -1,10 +1,9 @@
// Zed provider — RSA keypair callback auth (NOT standard OAuth).
export default {
id: "zed",
priority: 10,
priority: 999,
alias: "zd",
uiAlias: "zd",
hidden: true,
display: {
name: "Zed",
icon: "code",
+11 -2
View File
@@ -62,8 +62,17 @@ const ANTHROPIC_BETA_BASE = [
const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"];
// Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them.
export function selectAnthropicBeta(model = "") {
const flags = [...ANTHROPIC_BETA_BASE];
// `redact-thinking` asks Anthropic to return signature-only thinking blocks, which
// is right for clients that never render thinking but blanks the summaries a
// client explicitly requested with `thinking.display: "summarized"`.
const ANTHROPIC_BETA_REDACT_THINKING = "redact-thinking-2026-02-12";
export function wantsThinkingSummaries(body) {
return body?.thinking?.display === "summarized";
}
export function selectAnthropicBeta(model = "", body = null) {
const flags = ANTHROPIC_BETA_BASE.filter((flag) => flag !== ANTHROPIC_BETA_REDACT_THINKING || !wantsThinkingSummaries(body));
if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT);
return flags.join(",");
}
+16 -3
View File
@@ -26,6 +26,7 @@ const FORMAT_LEVELS = {
qwen: L.base,
kimi: L.levelMax,
deepseek: L.hiMax,
commandcode: ["none", "low", "medium", "high", "xhigh", "max"],
minimax: L.onOff,
hunyuan: L.base,
step: L.base,
@@ -35,17 +36,29 @@ const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
const PATTERN_THINKING = [
{ provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS },
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
// codebuddy-cn per-model effort sets — read off the client picker (server-
// delivered supportedEfforts), 2026-08-30. Gateway uses thinkingFormat "openai"
// but rejects levels outside each model's set.
// DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max
// all 200 via output_config.effort; "none" is a 400 on the anthropic route
// (disable thinking instead). none kept for the picker = disable.
{ pattern: "*deepseek-v4.*", levels: ["none", "low", "medium", "high", "xhigh", "max"] },
// codebuddy-cn per-model effort sets — the server's product-config payload
// publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint
// accepts any level you send (probed none/minimal/low/medium/high/xhigh/max
// → all 200), but values outside a model's supportedEfforts are silently
// clamped, so the declared set stays authoritative for the picker. Models
// that publish no supportedEfforts (glm-5.1 / glm-5v-turbo / kimi-k2.x /
// kimi-k3-1 / minimax-m3) fall through to the openai format default.
{ provider: "codebuddy-cn", pattern: "glm-5.3*", levels: ["low", "high", "max"] },
{ provider: "codebuddy-cn", pattern: "glm-5.2", levels: ["high", "xhigh"] },
{ provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
{ provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] },
{ provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] },
// codebuddy-intl rides the same gateway catalog, so its deepseek levels match.
{ provider: "codebuddy-intl", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
];
// Returns valid thinking levels for a model, or null when the model has no reasoning.