merge: sync upstream v0.5.81 into MIBP fork
# Conflicts: # .gitignore # Dockerfile # open-sse/handlers/chatCore.js # open-sse/providers/registry/cline.js # open-sse/providers/registry/index.js # open-sse/services/usage.js # open-sse/utils/streamHandler.js # package.json # src/app/(dashboard)/dashboard/profile/page.js # src/app/(dashboard)/dashboard/providers/[id]/page.js
This commit is contained in:
@@ -7,6 +7,9 @@ import { createRequire } from "module";
|
||||
export const GEMINI_CLI_VERSION = PROVIDERS["gemini-cli"]?.cliVersion;
|
||||
export const GEMINI_CLI_API_CLIENT = PROVIDERS["gemini-cli"]?.apiClient;
|
||||
|
||||
// === Codex CLI === derive từ registry codex.transport
|
||||
export const CODEX_CLI_VERSION = PROVIDERS["codex"]?.cliVersion;
|
||||
|
||||
// Map Node arch to Gemini CLI arch string (x64/x86/arm64/...)
|
||||
function geminiCLIArch() {
|
||||
const a = arch();
|
||||
@@ -175,6 +178,11 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C
|
||||
// makes the backend flag the request and answer 429 Quota Exhausted.
|
||||
export const ANTIGRAVITY_PROMPT_REWRITES = [
|
||||
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
|
||||
{ from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." },
|
||||
// Claude Code prepends this line to its system prompt. The Claude-format translator strips it,
|
||||
// but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions)
|
||||
// pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED.
|
||||
{ from: /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/gim, to: "" },
|
||||
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
|
||||
];
|
||||
|
||||
|
||||
@@ -27,11 +27,14 @@ const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]);
|
||||
// ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only.
|
||||
function findModel(models, modelId, aliasOrId) {
|
||||
if (!models) return undefined;
|
||||
const found = models.find(m => m.id === modelId);
|
||||
const baseModelId = typeof modelId === "string"
|
||||
? modelId.replace(/\([^()]+\)\s*$/, "").trim()
|
||||
: modelId;
|
||||
const found = models.find(m => m.id === modelId || m.id === baseModelId);
|
||||
if (found) return found;
|
||||
if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined;
|
||||
const normalized = normalizeModelId(modelId);
|
||||
if (normalized === modelId) return undefined;
|
||||
const normalized = normalizeModelId(baseModelId);
|
||||
if (normalized === baseModelId) return undefined;
|
||||
return models.find(m => m.id === normalized);
|
||||
}
|
||||
|
||||
@@ -50,7 +53,7 @@ export function findModelName(aliasOrId, modelId) {
|
||||
}
|
||||
|
||||
export function getModelTargetFormat(aliasOrId, modelId) {
|
||||
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode") && isMuseSparkModel(modelId)) {
|
||||
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) {
|
||||
return FORMATS.OPENAI_RESPONSES;
|
||||
}
|
||||
const models = PROVIDER_MODELS[aliasOrId];
|
||||
|
||||
@@ -3,10 +3,11 @@ import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js";
|
||||
import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
|
||||
import { cleanJSONSchemaForAntigravity, normalizeGeminiContents } from "../translator/formats/gemini.js";
|
||||
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
|
||||
import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js";
|
||||
|
||||
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
|
||||
function sanitizeFunctionName(name) {
|
||||
@@ -187,9 +188,12 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
};
|
||||
}
|
||||
|
||||
const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" });
|
||||
const sessionId = toNumericSessionId(rawSessionId) || rawSessionId;
|
||||
|
||||
// ─── Standard (non-image) request ───
|
||||
// Fix contents for Claude models via Antigravity
|
||||
const contents = body.request?.contents?.map(c => {
|
||||
const rawContents = (body.request?.contents || []).map(c => {
|
||||
let role = c.role;
|
||||
// functionResponse must be role "user" for Claude models
|
||||
if (c.parts?.some(p => p.functionResponse)) {
|
||||
@@ -202,21 +206,33 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
return true;
|
||||
});
|
||||
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
|
||||
// don't persist thoughtSignature in their history, so backfill the default signature on any
|
||||
// functionCall part that arrives without one.
|
||||
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
|
||||
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
|
||||
return {
|
||||
...c, role,
|
||||
parts: needsBackfill
|
||||
? parts.map(p => (p.functionCall && !p.thoughtSignature)
|
||||
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
|
||||
: p)
|
||||
: parts,
|
||||
};
|
||||
}
|
||||
return c;
|
||||
// don't persist thoughtSignature in their history, so backfill from cache or default signature.
|
||||
// In parallel function calls, only the first call needs a signature; siblings stay unsigned.
|
||||
let firstFunctionCallSeen = false;
|
||||
const modifiedParts = parts?.map(p => {
|
||||
if (!p.functionCall) return p;
|
||||
const callId = p.functionCall.id;
|
||||
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId, body.model || model) : null;
|
||||
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
|
||||
firstFunctionCallSeen = true;
|
||||
if (callSig) {
|
||||
return { ...p, thoughtSignature: callSig };
|
||||
}
|
||||
if (p.thoughtSignature && !cachedSig) {
|
||||
// Unsigned sibling call
|
||||
const { thoughtSignature: _, ...rest } = p;
|
||||
return rest;
|
||||
}
|
||||
return p;
|
||||
});
|
||||
|
||||
return {
|
||||
...c,
|
||||
role,
|
||||
parts: modifiedParts || parts || [],
|
||||
};
|
||||
});
|
||||
const contents = normalizeGeminiContents(rawContents);
|
||||
|
||||
// Sanitize tool schemas and function names before sending to Antigravity.
|
||||
let tools = body.request?.tools;
|
||||
@@ -267,7 +283,7 @@ export class AntigravityExecutor extends BaseExecutor {
|
||||
generationConfig,
|
||||
...(contents && { contents }),
|
||||
...(tools && { tools }),
|
||||
sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }),
|
||||
sessionId,
|
||||
safetySettings: undefined,
|
||||
...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } })
|
||||
};
|
||||
|
||||
@@ -127,7 +127,7 @@ export class BaseExecutor {
|
||||
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
|
||||
const url = this.buildUrl(model, stream, urlIndex, credentials);
|
||||
const transformedBody = this.transformRequest(model, body, stream, credentials);
|
||||
const headers = this.buildHeaders(credentials, stream, url, model);
|
||||
const headers = this.buildHeaders(credentials, stream, url, model, transformedBody);
|
||||
|
||||
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@ import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
|
||||
import { dbg } from "../utils/debugLog.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { stripCodexUnsupportedPatterns } from "../utils/codexToolSchema.js";
|
||||
|
||||
// SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts.
|
||||
const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
|
||||
@@ -72,6 +73,9 @@ function stripStoredItemReferences(body) {
|
||||
function normalizeCodexTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
// Codex's schema validator has no Unicode property escapes; a `pattern`
|
||||
// carrying `\p{...}` 400s the whole request on every account (#3922).
|
||||
const patternStats = { removed: 0 };
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const type = typeof tool.type === "string" ? tool.type : "";
|
||||
@@ -80,6 +84,9 @@ function normalizeCodexTools(body) {
|
||||
for (const st of tool.tools) {
|
||||
const n = typeof st?.name === "string" ? st.name.trim().slice(0, 128) : "";
|
||||
if (n) validNames.add(n);
|
||||
if (st?.parameters && typeof st.parameters === "object") {
|
||||
st.parameters = stripCodexUnsupportedPatterns(st.parameters, patternStats);
|
||||
}
|
||||
}
|
||||
}
|
||||
return true;
|
||||
@@ -101,10 +108,13 @@ function normalizeCodexTools(body) {
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, 128);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
tool.parameters = stripCodexUnsupportedPatterns(parameters, patternStats);
|
||||
validNames.add(name);
|
||||
return true;
|
||||
});
|
||||
if (patternStats.removed > 0) {
|
||||
dbg("CODEX", `stripped ${patternStats.removed} unsupported tool schema pattern(s)`);
|
||||
}
|
||||
// Drop tool_choice if it references an unknown function name
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
|
||||
@@ -40,10 +40,24 @@ export class CommandCodeExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
async execute(opts) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
|
||||
return result;
|
||||
const maxRetries = 2;
|
||||
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||||
const result = await super.execute(opts);
|
||||
if (!result?.response?.ok || !result.response.body) return result;
|
||||
|
||||
const wrappedResponse = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
|
||||
if (!wrappedResponse.ok && attempt < maxRetries) {
|
||||
const isRetryableStatus = wrappedResponse.status === 502 || wrappedResponse.status === 503 || wrappedResponse.status === 504;
|
||||
if (isRetryableStatus) {
|
||||
opts.log?.debug?.("RETRY", `CommandCode upstream returned status ${wrappedResponse.status}, retrying ${attempt + 1}/${maxRetries}...`);
|
||||
await new Promise(r => setTimeout(r, 1000 * (attempt + 1)));
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
result.response = wrappedResponse;
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
parseError(response, bodyText) {
|
||||
|
||||
@@ -151,7 +151,7 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
return BEARER;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
buildHeaders(credentials, stream = true, url, model, body = null) {
|
||||
const rt = credentials?.runtimeTransport;
|
||||
const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) };
|
||||
const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor();
|
||||
@@ -159,8 +159,19 @@ export class DefaultExecutor extends BaseExecutor {
|
||||
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
|
||||
applyAuth(headers, desc, credentials);
|
||||
|
||||
if (this.provider === "claude" && model) {
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
|
||||
// anthropic-compatible-* nodes serving a real Claude model sit in front of
|
||||
// Anthropic itself (a rotating multi-account proxy, a corporate gateway),
|
||||
// so the request needs the same beta flags the `claude` provider sends:
|
||||
// without `context-management-2025-06-27` upstream rejects the
|
||||
// `context_management` block Claude Code puts in every request with
|
||||
// "context_management: Extra inputs are not permitted" (HTTP 400), and the
|
||||
// combo silently falls through to the next model. The model id gates this:
|
||||
// a node fronting Kimi or GLM answers on its own ids and never matches, so
|
||||
// gateways that would choke on unknown beta flags are left untouched.
|
||||
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
|
||||
if (model && (this.provider === "claude"
|
||||
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
|
||||
headers["Anthropic-Beta"] = selectAnthropicBeta(model, body);
|
||||
}
|
||||
|
||||
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams
|
||||
|
||||
@@ -10,12 +10,14 @@ import { CodexExecutor } from "./codex.js";
|
||||
import { CursorExecutor } from "./cursor.js";
|
||||
import { VertexExecutor } from "./vertex.js";
|
||||
import { OpenCodeExecutor } from "./opencode.js";
|
||||
import { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
import { GrokWebExecutor } from "./grok-web.js";
|
||||
import { GrokCliExecutor } from "./grok-cli.js";
|
||||
import { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
import { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
import { CommandCodeExecutor } from "./commandcode.js";
|
||||
import { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
|
||||
import { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
|
||||
import { MimoFreeExecutor } from "./mimo-free.js";
|
||||
import { CodeBuddyExecutor } from "./codebuddy-cn.js";
|
||||
import { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";
|
||||
@@ -41,6 +43,7 @@ const executors = {
|
||||
vertex: new VertexExecutor("vertex"),
|
||||
"vertex-partner": new VertexExecutor("vertex-partner"),
|
||||
opencode: new OpenCodeExecutor(),
|
||||
"opencode-go": new OpenCodeGoExecutor(),
|
||||
"grok-web": new GrokWebExecutor(),
|
||||
"grok-cli": new GrokCliExecutor(),
|
||||
gcli: new GrokCliExecutor(), // Alias
|
||||
@@ -49,6 +52,7 @@ const executors = {
|
||||
"ollama-local": new OllamaLocalExecutor(),
|
||||
commandcode: new CommandCodeExecutor(),
|
||||
"xiaomi-tokenplan": new XiaomiTokenplanExecutor(),
|
||||
"xiaomi-mimo": new XiaomiMimoExecutor(),
|
||||
"mimo-free": new MimoFreeExecutor(),
|
||||
mmf: new MimoFreeExecutor(), // Alias for mimo-free
|
||||
"codebuddy-cn": new CodeBuddyExecutor(),
|
||||
@@ -86,12 +90,14 @@ export { CursorExecutor } from "./cursor.js";
|
||||
export { VertexExecutor } from "./vertex.js";
|
||||
export { DefaultExecutor } from "./default.js";
|
||||
export { OpenCodeExecutor } from "./opencode.js";
|
||||
export { OpenCodeGoExecutor } from "./opencode-go.js";
|
||||
export { GrokWebExecutor } from "./grok-web.js";
|
||||
export { GrokCliExecutor } from "./grok-cli.js";
|
||||
export { PerplexityWebExecutor } from "./perplexity-web.js";
|
||||
export { OllamaLocalExecutor } from "./ollama-local.js";
|
||||
export { CommandCodeExecutor } from "./commandcode.js";
|
||||
export { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
|
||||
export { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
|
||||
export { MimoFreeExecutor } from "./mimo-free.js";
|
||||
export { CodeBuddyExecutor } from "./codebuddy-cn.js";
|
||||
export { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";
|
||||
|
||||
+36
-16
@@ -127,12 +127,18 @@ async function readResponsePrefix(response, signal, maxBytes, timeoutMs) {
|
||||
return decoder.decode(concatChunks(chunks, totalBytes));
|
||||
}
|
||||
|
||||
// The instruction goes into the current user turn, never into a top-level
|
||||
// `systemPrompt`: kiro.dev answers any body carrying that field with
|
||||
// 400 REQUEST_BODY_INVALID, so writing it here turned every repair retry into
|
||||
// a hard failure.
|
||||
function appendRepairInstruction(body, kind) {
|
||||
const repaired = structuredClone(body || {});
|
||||
const instruction = REPAIR_INSTRUCTIONS[kind] || "Retry the previous incomplete Kiro response.";
|
||||
repaired.systemPrompt = repaired.systemPrompt
|
||||
? `${repaired.systemPrompt}\n\n${instruction}`
|
||||
: instruction;
|
||||
const msg = repaired?.conversationState?.currentMessage?.userInputMessage;
|
||||
if (msg) {
|
||||
const content = typeof msg.content === "string" ? msg.content : "";
|
||||
msg.content = content ? `${content}\n\n${instruction}` : instruction;
|
||||
}
|
||||
return repaired;
|
||||
}
|
||||
|
||||
@@ -259,6 +265,19 @@ export class KiroExecutor extends BaseExecutor {
|
||||
}
|
||||
}
|
||||
|
||||
// CLIRO parity for the Amazon surfaces: the Kiro runtime accepts the
|
||||
// SSO bearer header + agent-mode marker. Without these the deprecated
|
||||
// path gateway answers REQUEST_BODY_INVALID for modern payloads.
|
||||
if (credentials?.accessToken) {
|
||||
headers["x-amz-sso-bearer"] = credentials.accessToken;
|
||||
}
|
||||
headers["x-amzn-kiro-agent-mode"] = "spec";
|
||||
headers["x-amzn-codewhisperer-machine-id"] = "kiro-desktop";
|
||||
const profileArn = credentials?.providerSpecificData?.profileArn;
|
||||
if (profileArn) {
|
||||
headers["x-amzn-codewhisperer-profile-arn"] = profileArn;
|
||||
}
|
||||
|
||||
return headers;
|
||||
}
|
||||
|
||||
@@ -285,9 +304,13 @@ export class KiroExecutor extends BaseExecutor {
|
||||
// 403 "bearer token invalid", so they must hit the CodeWhisperer
|
||||
// *.amazonaws.com surface, and in the region the token was minted in
|
||||
// (the baseUrls are hardcoded us-east-1).
|
||||
const isCodeWhispererSurface =
|
||||
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
|
||||
if (!isCodeWhispererSurface) return baseUrls;
|
||||
// Kiro deprecated the legacy path-style GenerateAssistantResponse on
|
||||
// runtime.*.kiro.dev (IDE 1.0.228+ moved to POST / + x-amz-target). The
|
||||
// path gateway now answers valid modern payloads with 400
|
||||
// REQUEST_BODY_INVALID, and 400 is terminal in BaseExecutor, so kiro.dev
|
||||
// must never be the first surface for any auth method. Amazon surfaces
|
||||
// reject foreign tokens with 401/403, which DO fall through, so trying
|
||||
// q/codewhisperer first is safe for every auth method (CLIRO parity).
|
||||
|
||||
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
|
||||
const regionalize = (u) =>
|
||||
@@ -297,20 +320,17 @@ export class KiroExecutor extends BaseExecutor {
|
||||
|
||||
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
|
||||
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
|
||||
if (authMethod === "api_key") {
|
||||
const q = amazon.filter((u) => u.includes("://q."));
|
||||
const remaining = amazon.filter((u) => !u.includes("://q."));
|
||||
return q.length > 0
|
||||
? [...q, ...remaining, ...others]
|
||||
: [...amazon, ...others];
|
||||
}
|
||||
|
||||
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
|
||||
const q = amazon.filter((u) => u.includes("://q."));
|
||||
const remaining = amazon.filter((u) => !u.includes("://q."));
|
||||
return q.length > 0
|
||||
? [...q, ...remaining, ...others]
|
||||
: [...amazon, ...others];
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
const baseUrls = this.getOrderedBaseUrls(credentials);
|
||||
return baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
|
||||
const url = baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
|
||||
return url;
|
||||
}
|
||||
|
||||
// Retry only endpoint/auth-surface failures. Payload-invalid HTTP 400 must be
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
import crypto from "node:crypto";
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { modelTargetFormat } from "../providers/models/schema.js";
|
||||
import { getProviderModels } from "../config/providerModels.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../translator/formats/responsesApi.js";
|
||||
|
||||
const SESSION_HEADER = "x-opencode-session";
|
||||
const SESSION_FIELD = "_opencodeGoSession";
|
||||
const MAX_SESSION_LENGTH = 256;
|
||||
|
||||
const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses";
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
|
||||
function normalizeSession(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function nativeSession(headers) {
|
||||
if (!headers || typeof headers !== "object") return null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function translatedSession(sessionId, clientTool) {
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
|
||||
.digest("hex")
|
||||
.slice(0, 32);
|
||||
return `ses_${digest}`;
|
||||
}
|
||||
|
||||
// Strip the thinking suffix "model(level)" so checks hit the base id.
|
||||
function baseModelId(model) {
|
||||
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
|
||||
}
|
||||
|
||||
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …).
|
||||
// Reading the registry keeps this in sync with config — never hardcode model ids here.
|
||||
function isResponsesModel(model) {
|
||||
const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model));
|
||||
return modelTargetFormat(entry) === "openai-responses";
|
||||
}
|
||||
|
||||
// Flatten Chat Completions tool declarations into the Responses flat shape and
|
||||
// drop hosted/nameless tools the /responses endpoint rejects.
|
||||
function normalizeResponsesTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
|
||||
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
|
||||
? tool.parameters
|
||||
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
|
||||
// Mirror the request translator: {type:"object"} without properties is rejected
|
||||
// by strict Responses backends, so fill in the empty properties map.
|
||||
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
|
||||
if (!n || !validNames.has(n)) delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Last line of defense for native Responses clients (sourceFormat === targetFormat
|
||||
// skips translation): coerce items in place so malformed tool payloads 400 here
|
||||
// with a clear shape instead of upstream as InputValidationError.
|
||||
function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
// Strip prior-turn reasoning items: Muse Spark contributor models route to
|
||||
// an upstream Console backend where encrypted_content cannot be validated across
|
||||
// rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller".
|
||||
if (item.type === "reasoning") return false;
|
||||
delete item.encrypted_content;
|
||||
delete item.reasoning_encrypted_content;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.arguments = coerceResponsesArguments(item.arguments);
|
||||
return true;
|
||||
}
|
||||
if (item.type === "function_call_output") {
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.output = coerceResponsesOutput(item.output);
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
export class OpenCodeGoExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("opencode-go");
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
|
||||
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
|
||||
return super.buildUrl(model, stream, urlIndex, credentials);
|
||||
}
|
||||
|
||||
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
|
||||
const sourceCredentials = credentials || {};
|
||||
const native = nativeSession(sourceCredentials.rawHeaders);
|
||||
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
|
||||
headers: sourceCredentials.rawHeaders,
|
||||
body,
|
||||
connectionId: sourceCredentials.connectionId,
|
||||
scope: "opencode-go",
|
||||
});
|
||||
|
||||
return {
|
||||
...sourceCredentials,
|
||||
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
|
||||
};
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
const credentials = this.prepareRequestCredentials(args);
|
||||
return super.execute({ ...args, credentials });
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
const headers = super.buildHeaders(credentials || {}, stream, url, model);
|
||||
const prepared = credentials?.[SESSION_FIELD];
|
||||
if (prepared) {
|
||||
headers[SESSION_HEADER] = prepared;
|
||||
return headers;
|
||||
}
|
||||
|
||||
const fallback = this.prepareRequestCredentials({ credentials });
|
||||
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
|
||||
return headers;
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
const out = super.transformRequest(model, body);
|
||||
if (!isResponsesModel(model || body?.model)) return out;
|
||||
const normalized = normalizeResponsesInput(out.input);
|
||||
if (normalized) out.input = normalized;
|
||||
if (!Array.isArray(out.input) || out.input.length === 0) {
|
||||
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
|
||||
}
|
||||
// Responses names the output cap max_output_tokens, not max_tokens.
|
||||
if (out.max_output_tokens === undefined) {
|
||||
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
|
||||
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
|
||||
}
|
||||
delete out.max_tokens;
|
||||
delete out.max_completion_tokens;
|
||||
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
|
||||
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
|
||||
}
|
||||
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
|
||||
if (!out.reasoning.summary) out.reasoning.summary = "auto";
|
||||
}
|
||||
delete out.reasoning_effort;
|
||||
out.stream = true;
|
||||
out.store = false;
|
||||
normalizeResponsesTools(out);
|
||||
sanitizeResponsesItems(out);
|
||||
return out;
|
||||
}
|
||||
}
|
||||
+456
-25
@@ -1,24 +1,311 @@
|
||||
import crypto from "crypto";
|
||||
import { BaseExecutor } from "./base.js";
|
||||
import { PROVIDERS } from "../config/providers.js";
|
||||
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
|
||||
import { getThinkingLevels } from "../providers/thinkingLevels.js";
|
||||
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { isMuseSparkModel } from "../providers/models/helpers.js";
|
||||
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
|
||||
import {
|
||||
normalizeResponsesInput,
|
||||
clampResponsesCallId,
|
||||
coerceResponsesArguments,
|
||||
coerceResponsesOutput,
|
||||
} from "../translator/formats/responsesApi.js";
|
||||
|
||||
const OPENCODE_UA = "opencode";
|
||||
const OPENCODE_UA = "opencode/1.18.31";
|
||||
const MAX_SESSION_LENGTH = 256;
|
||||
const MAX_TOOL_NAME_LEN = 128;
|
||||
const SESSION_HEADER = "x-opencode-session";
|
||||
const SESSION_FIELD = "_opencodeSession";
|
||||
const REQ_FIELD = "_opencodeRequest";
|
||||
export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
|
||||
export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
|
||||
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
|
||||
|
||||
// OpenCode free tier requires both 'bash' and 'read' in tools payload.
|
||||
// Injected as cloaked decoy tools so external CLI tools (e.g. Claude Code's Bash/Read)
|
||||
// take precedence while satisfying upstream verification.
|
||||
const OPENCODE_DECOY_CHAT_TOOLS = [
|
||||
{
|
||||
type: "function",
|
||||
function: {
|
||||
name: "bash",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "function",
|
||||
function: {
|
||||
name: "read",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
const OPENCODE_DECOY_RESPONSES_TOOLS = [
|
||||
{
|
||||
type: "function",
|
||||
name: "bash",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
{
|
||||
type: "function",
|
||||
name: "read",
|
||||
description: "This tool is currently unavailable and must not be used.",
|
||||
parameters: { type: "object", properties: {} },
|
||||
},
|
||||
];
|
||||
|
||||
function cloakOpencodeTools(body, isResponses) {
|
||||
if (!body || typeof body !== "object") return;
|
||||
if (isResponses) {
|
||||
if (!Array.isArray(body.tools)) body.tools = [];
|
||||
const names = new Set(body.tools.map((t) => t.name || t.function?.name));
|
||||
for (const tool of OPENCODE_DECOY_RESPONSES_TOOLS) {
|
||||
if (!names.has(tool.name)) body.tools.push({ ...tool });
|
||||
}
|
||||
if (!body.tool_choice) body.tool_choice = "auto";
|
||||
} else {
|
||||
const hasTools = Array.isArray(body.tools) && body.tools.length > 0;
|
||||
if (!hasTools) {
|
||||
body.tools = OPENCODE_DECOY_CHAT_TOOLS.map((t) => ({ ...t, function: { ...t.function } }));
|
||||
if (!body.tool_choice) body.tool_choice = "none";
|
||||
} else {
|
||||
const names = new Set(body.tools.map((t) => t.function?.name || t.name));
|
||||
for (const tool of OPENCODE_DECOY_CHAT_TOOLS) {
|
||||
if (!names.has(tool.function.name)) {
|
||||
body.tools.push({ ...tool, function: { ...tool.function } });
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function hasValidOpencodeVersion(ua) {
|
||||
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
|
||||
if (!m) return false;
|
||||
const major = parseInt(m[1], 10);
|
||||
const minor = parseInt(m[2], 10);
|
||||
return major > 1 || (major === 1 && minor >= 17);
|
||||
}
|
||||
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
|
||||
const RESPONSES_MODELS = new Set([
|
||||
"muse-spark-1.2-contributor-free",
|
||||
"muse-spark-1.3-contributor-free",
|
||||
]);
|
||||
const MESSAGES_MODELS = new Set(["union-alpha"]);
|
||||
|
||||
function generateRequestId() {
|
||||
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
let lastTimestamp = 0;
|
||||
let counter = 0;
|
||||
|
||||
function unstableRandom() {
|
||||
const bytes = crypto.randomBytes(14);
|
||||
let randomPart = "";
|
||||
for (let i = 0; i < 14; i++) {
|
||||
randomPart += BASE62_CHARS[bytes[i] % 62];
|
||||
}
|
||||
return randomPart;
|
||||
}
|
||||
|
||||
function generateSessionId() {
|
||||
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
|
||||
export function generateSessionId(timestamp = Date.now()) {
|
||||
if (timestamp !== lastTimestamp) {
|
||||
lastTimestamp = timestamp;
|
||||
counter = 0;
|
||||
}
|
||||
counter++;
|
||||
|
||||
const current = BigInt(timestamp) * 0x1000n + BigInt(counter);
|
||||
const value = ~current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `ses_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function generateRequestId(timestamp = Date.now()) {
|
||||
const current = BigInt(timestamp) * 0x1000n + 1n;
|
||||
const value = current;
|
||||
const time = Array.from({ length: 6 }, (_, index) =>
|
||||
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
|
||||
.toString(16)
|
||||
.padStart(2, "0")
|
||||
).join("");
|
||||
return `msg_${time}${unstableRandom()}`;
|
||||
}
|
||||
|
||||
export function translateSessionId(sessionId, clientTool = "") {
|
||||
if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) {
|
||||
return sessionId.trim();
|
||||
}
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`)
|
||||
.digest();
|
||||
const timeHex = digest.subarray(0, 6).toString("hex");
|
||||
let randomPart = "";
|
||||
for (let i = 6; i < 20; i++) {
|
||||
randomPart += BASE62_CHARS[digest[i] % 62];
|
||||
}
|
||||
return `ses_${timeHex}${randomPart}`;
|
||||
}
|
||||
|
||||
function normalizeSession(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function nativeSession(headers) {
|
||||
if (!headers || typeof headers !== "object") return null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) {
|
||||
const normalized = normalizeSession(value);
|
||||
if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// Upstream free-tier quota is accounted per session. Minting a fresh
|
||||
// x-opencode-session on every request burns through it and surfaces as
|
||||
// 429 FreeUsageLimitError with growing reset-after delays, while the real
|
||||
// CLI reuses one long-lived canonical session per conversation. Mirror
|
||||
// that: one stable canonical session per downstream identity, evicted
|
||||
// after MEMORY_CONFIG.sessionTtlMs like the other session stores.
|
||||
const stableOpencodeSessions = new Map();
|
||||
const MAX_STABLE_SESSIONS = 1000;
|
||||
const stableSessionCleanup = setInterval(() => {
|
||||
const now = Date.now();
|
||||
for (const [key, entry] of stableOpencodeSessions) {
|
||||
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) {
|
||||
stableOpencodeSessions.delete(key);
|
||||
}
|
||||
}
|
||||
}, MEMORY_CONFIG.sessionCleanupIntervalMs);
|
||||
if (stableSessionCleanup.unref) stableSessionCleanup.unref();
|
||||
|
||||
function identityKey(credentials) {
|
||||
const connectionId = credentials?.connectionId || credentials?.id;
|
||||
if (connectionId) return `opencode:conn:${String(connectionId).slice(0, 128)}`;
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const auth = raw.authorization || raw.Authorization || raw["x-api-key"] || raw["X-Api-Key"] || "";
|
||||
if (auth) {
|
||||
const digest = crypto.createHash("sha256").update(String(auth)).digest("hex").slice(0, 32);
|
||||
return `opencode:auth:${digest}`;
|
||||
}
|
||||
return "opencode:default";
|
||||
}
|
||||
|
||||
export function stableSessionId(credentials) {
|
||||
const key = identityKey(credentials);
|
||||
const existing = stableOpencodeSessions.get(key);
|
||||
if (existing) {
|
||||
existing.lastUsed = Date.now();
|
||||
stableOpencodeSessions.delete(key);
|
||||
stableOpencodeSessions.set(key, existing);
|
||||
return existing.sessionId;
|
||||
}
|
||||
const sessionId = generateSessionId();
|
||||
if (stableOpencodeSessions.size >= MAX_STABLE_SESSIONS) {
|
||||
stableOpencodeSessions.delete(stableOpencodeSessions.keys().next().value);
|
||||
}
|
||||
stableOpencodeSessions.set(key, { sessionId, lastUsed: Date.now() });
|
||||
return sessionId;
|
||||
}
|
||||
|
||||
function lastUserText(body) {
|
||||
try {
|
||||
if (!body || typeof body !== "object") return "";
|
||||
const arr = Array.isArray(body.messages)
|
||||
? body.messages
|
||||
: Array.isArray(body.input)
|
||||
? body.input
|
||||
: null;
|
||||
if (!arr) return typeof body.input === "string" ? body.input.slice(-600) : "";
|
||||
for (let i = arr.length - 1; i >= 0; i--) {
|
||||
const msg = arr[i];
|
||||
if (!msg) continue;
|
||||
if (msg.role && msg.role !== "user") continue;
|
||||
const content = msg.content;
|
||||
if (typeof content === "string" && content.trim()) return content.trim().slice(-600);
|
||||
if (Array.isArray(content)) {
|
||||
const text = content
|
||||
.map((part) => (typeof part === "string" ? part : part?.text || part?.input_text || ""))
|
||||
.join(" ")
|
||||
.trim();
|
||||
if (text) return text.slice(-600);
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
return "";
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
// The real CLI sends the current user message id (stable per turn, same on
|
||||
// retries) as x-opencode-request. Derive it deterministically from the
|
||||
// session plus the last user message so retries share the id.
|
||||
export function deriveRequestId(sessionId, body) {
|
||||
const text = lastUserText(body);
|
||||
if (!text) return generateRequestId();
|
||||
const digest = crypto
|
||||
.createHash("sha256")
|
||||
.update(`opencode-req\0${sessionId || ""}\0${text}`)
|
||||
.digest();
|
||||
const timeHex = digest.subarray(0, 6).toString("hex");
|
||||
let randomPart = "";
|
||||
for (let i = 6; i < 20; i++) {
|
||||
randomPart += BASE62_CHARS[digest[i] % 62];
|
||||
}
|
||||
const id = `msg_${timeHex}${randomPart}`;
|
||||
return OPENCODE_REQUEST_RE.test(id) ? id : generateRequestId();
|
||||
}
|
||||
|
||||
function normalizeRequestId(value) {
|
||||
if (typeof value !== "string") return null;
|
||||
const normalized = value.trim();
|
||||
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
|
||||
return OPENCODE_REQUEST_RE.test(normalized) ? normalized : null;
|
||||
}
|
||||
|
||||
function bodyHasSessionHints(body) {
|
||||
try {
|
||||
if (!body || typeof body !== "object") return false;
|
||||
if (typeof body.session_id === "string" && body.session_id.trim()) return true;
|
||||
if (typeof body.conversation_id === "string" && body.conversation_id.trim()) return true;
|
||||
if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key.trim()) return true;
|
||||
if (body.metadata && typeof body.metadata.user_id === "string" && body.metadata.user_id.trim()) return true;
|
||||
if (body.request && body.request.sessionId != null && String(body.request.sessionId) !== "") return true;
|
||||
const arr = Array.isArray(body.messages)
|
||||
? body.messages
|
||||
: Array.isArray(body.input)
|
||||
? body.input
|
||||
: null;
|
||||
if (arr) {
|
||||
let assistantText = "";
|
||||
for (const msg of arr) {
|
||||
if (msg?.role === "assistant") {
|
||||
const content = msg.content;
|
||||
if (typeof content === "string") assistantText += content;
|
||||
else if (Array.isArray(content)) {
|
||||
for (const part of content) assistantText += part?.text || part?.output || "";
|
||||
}
|
||||
if (assistantText.length >= 50) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
|
||||
@@ -31,14 +318,116 @@ function isResponsesModel(model) {
|
||||
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials) {
|
||||
function isMessagesModel(model) {
|
||||
return MESSAGES_MODELS.has(baseModelId(model));
|
||||
}
|
||||
|
||||
function resolveOpencodeSession(body, credentials, providerSessionId, clientTool) {
|
||||
const headers = credentials?.rawHeaders || {};
|
||||
return resolveSessionId({
|
||||
headers,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
generate: generateSessionId,
|
||||
const native = nativeSession(headers);
|
||||
if (native) return native;
|
||||
|
||||
let incoming = null;
|
||||
for (const [key, value] of Object.entries(headers)) {
|
||||
if (key.toLowerCase() === SESSION_HEADER) {
|
||||
incoming = normalizeSession(value);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
const hinted = incoming || normalizeSession(providerSessionId);
|
||||
if (hinted) return translateSessionId(hinted, clientTool);
|
||||
|
||||
if (credentials?.connectionId || bodyHasSessionHints(body)) {
|
||||
let viaManager = null;
|
||||
try {
|
||||
viaManager = resolveSessionId({
|
||||
headers,
|
||||
body,
|
||||
connectionId: credentials?.connectionId,
|
||||
scope: "opencode",
|
||||
});
|
||||
} catch {
|
||||
viaManager = null;
|
||||
}
|
||||
if (viaManager) return translateSessionId(viaManager, clientTool);
|
||||
}
|
||||
|
||||
return stableSessionId(credentials);
|
||||
}
|
||||
|
||||
function resolveOpencodeRequestId(body, credentials, sessionId) {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
for (const [key, value] of Object.entries(raw)) {
|
||||
if (key.toLowerCase() === "x-opencode-request") {
|
||||
const normalized = normalizeRequestId(value);
|
||||
if (normalized) return normalized;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return deriveRequestId(sessionId, body);
|
||||
}
|
||||
|
||||
function normalizeResponsesTools(body) {
|
||||
if (!Array.isArray(body.tools)) return;
|
||||
const validNames = new Set();
|
||||
body.tools = body.tools.filter((tool) => {
|
||||
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
|
||||
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
|
||||
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
|
||||
const name = rawName.trim();
|
||||
if (!name) return false;
|
||||
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
|
||||
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
|
||||
? tool.parameters
|
||||
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
|
||||
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
|
||||
for (const k of Object.keys(tool)) delete tool[k];
|
||||
tool.type = "function";
|
||||
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
|
||||
if (description) tool.description = description;
|
||||
tool.parameters = parameters;
|
||||
validNames.add(tool.name);
|
||||
return true;
|
||||
});
|
||||
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
|
||||
if (body.tool_choice.type === "function") {
|
||||
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
|
||||
if (!n || !validNames.has(n)) delete body.tool_choice;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function sanitizeResponsesItems(body) {
|
||||
if (!Array.isArray(body.input)) return;
|
||||
body.input = body.input.filter((item) => {
|
||||
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
|
||||
// Strip prior-turn reasoning items: OpenCode Free uses public/pooled credentials
|
||||
// (`Bearer public`) routing to an upstream OpenAI/Console account pool.
|
||||
// OpenAI Responses API strictly enforces that reasoning `encrypted_content`
|
||||
// can only be decrypted by the exact caller/account that issued it; sending it
|
||||
// across different accounts or rotating proxy relays triggers:
|
||||
// [invalid_request_error] reasoning `encrypted_content` was not issued to this caller (400).
|
||||
// Furthermore, under stateless mode (store=false), omitting encrypted_content
|
||||
// causes OpenAI to reject the referenced reasoning item as "not found or was deleted".
|
||||
// Dropping prior reasoning items allows multi-turn conversations and tool-calling
|
||||
// loops to succeed cleanly.
|
||||
if (item.type === "reasoning") return false;
|
||||
delete item.encrypted_content;
|
||||
delete item.reasoning_encrypted_content;
|
||||
if (item.type === "function_call") {
|
||||
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
|
||||
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.arguments = coerceResponsesArguments(item.arguments);
|
||||
return true;
|
||||
}
|
||||
if (item.type === "function_call_output") {
|
||||
item.call_id = clampResponsesCallId(item.call_id);
|
||||
item.output = coerceResponsesOutput(item.output);
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -74,12 +463,35 @@ const IP_LIMIT_BODY = /limit|rate|quota|exhausted|capacity|too many|retry/i;
|
||||
export class OpenCodeExecutor extends BaseExecutor {
|
||||
constructor() {
|
||||
super("opencode", PROVIDERS.opencode);
|
||||
this._currentSessionId = null;
|
||||
}
|
||||
|
||||
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
|
||||
const sourceCredentials = credentials || {};
|
||||
const session = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool);
|
||||
|
||||
return {
|
||||
...sourceCredentials,
|
||||
[SESSION_FIELD]: session,
|
||||
[REQ_FIELD]: resolveOpencodeRequestId(body, sourceCredentials, session),
|
||||
};
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
this._currentSessionId = resolveOpencodeSession(body, credentials);
|
||||
if (isResponsesModel(model)) {
|
||||
if (body && typeof body === "object" && model && !body.model) body.model = model;
|
||||
// Zen rejects non-streaming requests on free models with 403 FreeTierError;
|
||||
// always stream upstream and let the handler layer aggregate for non-stream clients.
|
||||
if (body && typeof body === "object") body.stream = true;
|
||||
if (isResponsesModel(model || body?.model) && body && typeof body === "object") {
|
||||
// ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng.
|
||||
if ("tool_choice" in body && body.tool_choice !== "auto"
|
||||
&& this.config.quirks?.forceAutoToolChoiceModels?.includes(baseModelId(model))) {
|
||||
body.tool_choice = "auto";
|
||||
}
|
||||
const normalized = normalizeResponsesInput(body.input);
|
||||
if (normalized) body.input = normalized;
|
||||
if (!Array.isArray(body.input) || body.input.length === 0) {
|
||||
body.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
|
||||
}
|
||||
// Responses API names the output cap max_output_tokens and takes thinking
|
||||
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
|
||||
if (body.max_output_tokens === undefined) {
|
||||
@@ -89,35 +501,54 @@ export class OpenCodeExecutor extends BaseExecutor {
|
||||
delete body.max_tokens;
|
||||
delete body.max_completion_tokens;
|
||||
normalizeOpencodeReasoning(model, body);
|
||||
body.stream = true;
|
||||
body.store = false;
|
||||
normalizeResponsesTools(body);
|
||||
sanitizeResponsesItems(body);
|
||||
if (!Array.isArray(body.tools) || body.tools.length === 0) {
|
||||
cloakOpencodeTools(body, true);
|
||||
}
|
||||
} else if (body && typeof body === "object") {
|
||||
cloakOpencodeTools(body, false);
|
||||
}
|
||||
return injectReasoningContent({ provider: this.provider, model, body });
|
||||
}
|
||||
|
||||
buildUrl(model) {
|
||||
const base = this.config.baseUrl;
|
||||
return isResponsesModel(model)
|
||||
? `${base}/zen/v1/responses`
|
||||
: `${base}/zen/v1/chat/completions`;
|
||||
async execute(args) {
|
||||
return super.execute({ ...args, credentials: this.prepareRequestCredentials(args) });
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true) {
|
||||
buildUrl(model) {
|
||||
const base = this.config.baseUrl;
|
||||
if (isResponsesModel(model)) return `${base}/zen/v1/responses`;
|
||||
if (isMessagesModel(model)) return `${base}/zen/v1/messages`;
|
||||
return `${base}/zen/v1/chat/completions`;
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url = "") {
|
||||
const raw = credentials?.rawHeaders || {};
|
||||
const lower = {};
|
||||
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
|
||||
|
||||
const downstreamUa = lower["user-agent"] || "";
|
||||
const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode");
|
||||
const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa);
|
||||
|
||||
return {
|
||||
const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD];
|
||||
const downstreamReq = normalizeRequestId(lower["x-opencode-request"]);
|
||||
const requestId = credentials?.[REQ_FIELD] || downstreamReq || generateRequestId();
|
||||
|
||||
const headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": "Bearer public",
|
||||
"User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA,
|
||||
"x-opencode-client": lower["x-opencode-client"] || "desktop",
|
||||
"x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(),
|
||||
"x-opencode-request": lower["x-opencode-request"] || generateRequestId(),
|
||||
"x-opencode-session": session,
|
||||
"x-opencode-request": requestId,
|
||||
"x-opencode-project": lower["x-opencode-project"] || "global",
|
||||
"Accept": stream ? "text/event-stream" : "*/*",
|
||||
};
|
||||
if (url.endsWith("/messages")) headers["anthropic-version"] = ANTHROPIC_API_VERSION;
|
||||
return headers;
|
||||
}
|
||||
|
||||
parseError(response, bodyText) {
|
||||
|
||||
+150
-31
@@ -31,16 +31,21 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js";
|
||||
import { SSE_DONE } from "../utils/sseConstants.js";
|
||||
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
|
||||
import {
|
||||
QODER_CHAT_URL_ENCODED,
|
||||
QODER_CHAT_BASE_ALT,
|
||||
QODER_CHAT_SIG_PATH,
|
||||
QODER_MODEL_MAP,
|
||||
QODER_CONTEXT_TIER_ENV,
|
||||
qoderInferenceBase,
|
||||
} from "../shared/qoder/constants.js";
|
||||
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
|
||||
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js";
|
||||
import { encodeDataUri } from "../translator/concerns/image.js";
|
||||
import { createQoderSseCoalescer } from "../shared/qoder/sse.js";
|
||||
import { rewriteQoderMessageAttachments } from "../shared/qoder/attachments.js";
|
||||
import { resolveQoderContextTier, applyQoderContextTier } from "../shared/qoder/contextTier.js";
|
||||
|
||||
/**
|
||||
* Hoist role:"system" messages out of the messages array (Qoder rejects
|
||||
* system in messages) and flatten any multipart content arrays.
|
||||
* system in messages) and flatten multipart content arrays — EXCEPT image
|
||||
* blocks, which are preserved (see normalizeContent).
|
||||
*/
|
||||
function normalizeMessages(messages) {
|
||||
if (!Array.isArray(messages) || messages.length === 0) {
|
||||
@@ -50,18 +55,88 @@ function normalizeMessages(messages) {
|
||||
const out = [];
|
||||
for (const msg of messages) {
|
||||
if (!msg || typeof msg !== "object") continue;
|
||||
const text = extractText(msg.content);
|
||||
if (msg.role === "system") {
|
||||
const text = extractText(msg.content);
|
||||
if (text) systemParts.push(text);
|
||||
continue;
|
||||
}
|
||||
const cloned = { ...msg };
|
||||
cloned.content = text;
|
||||
cloned.content = normalizeContent(msg.content);
|
||||
out.push(cloned);
|
||||
}
|
||||
return { messages: out, systemText: systemParts.join("\n\n") };
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize one message's content for Qoder.
|
||||
*
|
||||
* Text-only content is flattened to a plain string (Qoder's historical
|
||||
* shape). When images are present the content stays an array and image
|
||||
* blocks are kept as OpenAI-style `image_url` parts. Native qodercli
|
||||
* uploads inlined bytes to `/api/v2/image/upload` first and then sends
|
||||
* the OSS URL — `buildQoderRequestBody` does that rewrite before this
|
||||
* runs. Tiny leftover data URIs are still accepted. The legacy
|
||||
* top-level `image_urls` / `chat_context.imageUrls` slots stay null —
|
||||
* qodercli leaves them null too.
|
||||
*
|
||||
* Claude-style `{type:"image", source:{...}}` blocks are converted to
|
||||
* `image_url`. File/document blocks that survived rewrite become short
|
||||
* stubs so 30MB PDFs never land in agent_chat_generation.
|
||||
*/
|
||||
function normalizeContent(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (content == null) return "";
|
||||
if (!Array.isArray(content)) return String(content);
|
||||
|
||||
const blocks = [];
|
||||
const textParts = [];
|
||||
let hasImage = false;
|
||||
|
||||
const pushText = (text) => {
|
||||
if (!text) return;
|
||||
if (hasImage || blocks.length) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
|
||||
else textParts.push(text);
|
||||
};
|
||||
|
||||
const imageUrlOf = (item) => {
|
||||
if (typeof item.image_url === "string" && item.image_url) return item.image_url;
|
||||
if (typeof item.image_url?.url === "string" && item.image_url.url) return item.image_url.url;
|
||||
return null;
|
||||
};
|
||||
|
||||
for (const item of content) {
|
||||
if (!item || typeof item !== "object") continue;
|
||||
const imageUrl = item.type === OPENAI_BLOCK.IMAGE_URL ? imageUrlOf(item) : null;
|
||||
if (imageUrl) {
|
||||
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: imageUrl } });
|
||||
hasImage = true;
|
||||
} else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) {
|
||||
// Claude base64/url image → OpenAI image_url equivalent.
|
||||
const src = item.source;
|
||||
const url = src.type === "base64" && src.data
|
||||
? encodeDataUri(src.media_type || "image/png", src.data)
|
||||
: typeof src.url === "string" && src.url ? src.url : null;
|
||||
if (url) {
|
||||
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } });
|
||||
hasImage = true;
|
||||
}
|
||||
} else if (item.type === OPENAI_BLOCK.FILE) {
|
||||
const name = item.file?.filename || item.file?.name || "file";
|
||||
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
|
||||
} else if (item.type === CLAUDE_BLOCK.DOCUMENT) {
|
||||
const name = item.title || "document";
|
||||
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
|
||||
} else if (typeof item.text === "string" && item.text) {
|
||||
pushText(item.text);
|
||||
}
|
||||
}
|
||||
|
||||
if (!hasImage) return textParts.join("\n");
|
||||
// Prepend any text collected before the first image block.
|
||||
if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") });
|
||||
return blocks;
|
||||
}
|
||||
|
||||
function extractText(content) {
|
||||
if (typeof content === "string") return content;
|
||||
if (content == null) return "";
|
||||
@@ -84,9 +159,9 @@ function extractText(content) {
|
||||
function lastUserText(messages) {
|
||||
for (let i = messages.length - 1; i >= 0; i--) {
|
||||
const m = messages[i];
|
||||
if (m?.role === "user" && typeof m.content === "string") {
|
||||
return m.content;
|
||||
}
|
||||
if (m?.role !== "user") continue;
|
||||
if (typeof m.content === "string") return m.content;
|
||||
if (Array.isArray(m.content)) return extractText(m.content);
|
||||
}
|
||||
return "";
|
||||
}
|
||||
@@ -110,6 +185,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) {
|
||||
if (m.role) { h.update("\0"); h.update(m.role); }
|
||||
if (typeof m.content === "string" && m.content) {
|
||||
h.update("\0"); h.update(m.content);
|
||||
} else if (Array.isArray(m.content)) {
|
||||
// Include image refs so the same prompt with a different image gets
|
||||
// a distinct chat_record_id.
|
||||
h.update("\0");
|
||||
try { h.update(JSON.stringify(m.content)); } catch {}
|
||||
}
|
||||
}
|
||||
if (tools) {
|
||||
@@ -127,7 +207,7 @@ function truncate(s, n) {
|
||||
/**
|
||||
* Map the OpenAI-style request body into the exact shape Qoder expects.
|
||||
*/
|
||||
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }) {
|
||||
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null }) {
|
||||
const qoderKey = String(model || "").replace(/^qoder\//, "");
|
||||
|
||||
// Fetch model config from dynamic API instead of relying on static QODER_MODEL_MAP.
|
||||
@@ -146,7 +226,30 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
modelConfig = { ...retried, key: qoderKey };
|
||||
}
|
||||
|
||||
const { messages, systemText } = normalizeMessages(body.messages || []);
|
||||
const incoming = Array.isArray(body.messages)
|
||||
? body.messages.map((m) => {
|
||||
if (!m || typeof m !== "object") return m;
|
||||
return {
|
||||
...m,
|
||||
content: Array.isArray(m.content)
|
||||
? m.content.map((b) => (b && typeof b === "object" ? { ...b } : b))
|
||||
: m.content,
|
||||
};
|
||||
})
|
||||
: [];
|
||||
try {
|
||||
await rewriteQoderMessageAttachments(incoming, {
|
||||
credentials,
|
||||
log,
|
||||
proxyOptions,
|
||||
signal,
|
||||
uploadFn,
|
||||
});
|
||||
} catch (err) {
|
||||
log?.warn?.("QODER", `attachment rewrite failed: ${err.message}`);
|
||||
}
|
||||
|
||||
const { messages, systemText } = normalizeMessages(incoming);
|
||||
const tools = body.tools;
|
||||
const isReasoning = !!modelConfig.is_reasoning;
|
||||
const maxOutputTokens = Number(modelConfig.max_output_tokens) || 0;
|
||||
@@ -165,7 +268,21 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
const sessionId = stableHash("qoder-session", psd.userId, qoderKey);
|
||||
const recordId = stableChatRecordId(qoderKey, messages, tools, maxTokens);
|
||||
|
||||
return {
|
||||
// Context-window tier (200K/400K/1M): the IDE picks one from model_config.context_config;
|
||||
// qodercli-style requests default to the smallest. Escalate when the prompt no longer fits.
|
||||
const tierChoice = resolveQoderContextTier(
|
||||
modelConfig,
|
||||
{ system: systemText, messages, tools },
|
||||
{ preference: process.env[QODER_CONTEXT_TIER_ENV] },
|
||||
);
|
||||
if (tierChoice) {
|
||||
log?.info?.(
|
||||
"QODER",
|
||||
`context tier ${tierChoice.tier.name} (${tierChoice.tier.tokenCount} tokens, ${tierChoice.reason}) for ~${tierChoice.estimatedTokens} prompt tokens`,
|
||||
);
|
||||
}
|
||||
|
||||
const built = {
|
||||
qoderKey,
|
||||
payload: {
|
||||
request_id: uuidv4(),
|
||||
@@ -213,6 +330,8 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
|
||||
},
|
||||
modelConfig,
|
||||
};
|
||||
if (tierChoice) applyQoderContextTier(built.payload, tierChoice.tier);
|
||||
return built;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -276,6 +395,11 @@ async function peekFirstQoderFrame(reader, decoder) {
|
||||
* response.text() which hangs until the socket closes — so on terminal
|
||||
* events we cancel the upstream reader and close our stream immediately.
|
||||
*
|
||||
* Usage: Qoder puts finish_reason on `delta` and sends token counts on a
|
||||
* later `choices: []` frame. Downstream OpenAI/Claude clients only read
|
||||
* usage from the finish chunk, so we coalesce those two frames (see
|
||||
* createQoderSseCoalescer) before forwarding.
|
||||
*
|
||||
* NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl).
|
||||
* If detected, return 403 response so chatCore marks connection unavailable
|
||||
* and triggers combo fallback instead of leaking error text into chat.
|
||||
@@ -302,6 +426,11 @@ async function wrapQoderSSE(response, model) {
|
||||
const upstreamDrained = peek.upstreamDone === true;
|
||||
const encoder = new TextEncoder();
|
||||
let doneEmitted = false;
|
||||
const coalescer = createQoderSseCoalescer({ model, encoder, sseDone: SSE_DONE });
|
||||
|
||||
const syncDone = () => {
|
||||
if (coalescer.doneEmitted) doneEmitted = true;
|
||||
};
|
||||
|
||||
// Process one already-extracted SSE line (no trailing newline).
|
||||
const processLine = (line, controller) => {
|
||||
@@ -312,15 +441,17 @@ async function wrapQoderSSE(response, model) {
|
||||
|
||||
const data = trimmed.slice(5).trimStart();
|
||||
if (data === "[DONE]") {
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
doneEmitted = true;
|
||||
coalescer.flush(controller);
|
||||
syncDone();
|
||||
return;
|
||||
}
|
||||
|
||||
let envelope;
|
||||
try { envelope = JSON.parse(data); } catch { return; }
|
||||
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
|
||||
const inner = typeof envelope.body === "string" ? envelope.body : "";
|
||||
const inner = typeof envelope.body === "string"
|
||||
? envelope.body
|
||||
: envelope.body != null ? JSON.stringify(envelope.body) : "";
|
||||
if (statusVal !== 200) {
|
||||
const msg = inner || `upstream status ${statusVal}`;
|
||||
const errChunk = JSON.stringify({
|
||||
@@ -336,14 +467,8 @@ async function wrapQoderSSE(response, model) {
|
||||
return;
|
||||
}
|
||||
if (!inner) return;
|
||||
if (inner === "[DONE]") {
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
doneEmitted = true;
|
||||
return;
|
||||
}
|
||||
// Strip embedded newlines so the SSE frame stays a single event.
|
||||
const sanitized = inner.replace(/\r?\n/g, "");
|
||||
controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`));
|
||||
coalescer.handleInner(inner, controller);
|
||||
syncDone();
|
||||
};
|
||||
|
||||
const stream = new ReadableStream({
|
||||
@@ -402,7 +527,7 @@ async function wrapQoderSSE(response, model) {
|
||||
} finally {
|
||||
if (!doneEmitted) {
|
||||
try {
|
||||
controller.enqueue(encoder.encode(SSE_DONE));
|
||||
coalescer.flush(controller);
|
||||
doneEmitted = true;
|
||||
} catch { /* already closed */ }
|
||||
}
|
||||
@@ -431,13 +556,7 @@ export class QoderExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
buildUrl(credentials) {
|
||||
// Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt-
|
||||
// with "Login expired" (403). Device tokens (dt-...) stay on api3.
|
||||
const raw = credentials?.apiKey || credentials?.accessToken;
|
||||
if (typeof raw === "string" && !raw.startsWith("pt-") && (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))) {
|
||||
return `${QODER_CHAT_BASE_ALT}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
|
||||
}
|
||||
return QODER_CHAT_URL_ENCODED;
|
||||
return `${qoderInferenceBase(credentials)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
|
||||
}
|
||||
|
||||
// Override execute entirely — Qoder needs:
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
import { DefaultExecutor } from "./default.js";
|
||||
import { getMimoAccountCookie, invalidateMimoAccountCookieCache, MIMO_API_BASE, MIMO_API_UA } from "../shared/mimoAccount.js";
|
||||
|
||||
// Desktop-exclusive Preview models. These are served by the account service's
|
||||
// /api/route proxy, authorized by the Xiaomi account session (NOT the sk- key).
|
||||
// See shared/mimoAccount.js for the session handshake.
|
||||
const PREVIEW_MODELS = new Set(["mimo-x-pro-preview", "mimo-x-flash-preview"]);
|
||||
|
||||
// Session cookie resolved in execute() (async) and read back by buildHeaders()
|
||||
// (sync — BaseExecutor.execute does not await it). Carried on the per-request
|
||||
// credentials object, same as runtimeTransport.
|
||||
const COOKIE_KEY = "__mimoAccountCookie";
|
||||
|
||||
// Upstream calls may hand us either the bare id or a `provider/model` ref.
|
||||
function bareModel(model) {
|
||||
const s = String(model || "");
|
||||
const i = s.indexOf("/");
|
||||
return i >= 0 ? s.slice(i + 1) : s;
|
||||
}
|
||||
|
||||
export class XiaomiMimoExecutor extends DefaultExecutor {
|
||||
constructor() {
|
||||
super("xiaomi-mimo");
|
||||
}
|
||||
|
||||
static isPreviewModel(model) {
|
||||
return PREVIEW_MODELS.has(bareModel(model));
|
||||
}
|
||||
|
||||
buildUrl(model, stream, urlIndex = 0, credentials = null) {
|
||||
// Preview models live on the account-service route, which is not one of the
|
||||
// declared transports — resolve it before the default runtimeTransport path.
|
||||
if (XiaomiMimoExecutor.isPreviewModel(model)) {
|
||||
return `${MIMO_API_BASE}/api/route/chat/completions`;
|
||||
}
|
||||
// Cloud API models keep default handling, so a Claude-format client reaches
|
||||
// the /anthropic/v1/messages transport.
|
||||
return super.buildUrl(model, stream, urlIndex, credentials);
|
||||
}
|
||||
|
||||
buildHeaders(credentials, stream = true, url, model) {
|
||||
if (XiaomiMimoExecutor.isPreviewModel(model) && credentials?.[COOKIE_KEY]) {
|
||||
// Preview models authenticate with the account-session cookie, not the key.
|
||||
return {
|
||||
"Content-Type": "application/json",
|
||||
Accept: stream ? "text/event-stream" : "application/json",
|
||||
"User-Agent": MIMO_API_UA,
|
||||
Cookie: credentials[COOKIE_KEY],
|
||||
};
|
||||
}
|
||||
return super.buildHeaders(credentials, stream, url, model);
|
||||
}
|
||||
|
||||
transformRequest(model, body, stream, credentials) {
|
||||
// super runs stripUnsupportedParams, which flattens Preview content-part
|
||||
// arrays (see the xiaomi-mimo rule in translator/concerns/paramSupport.js).
|
||||
const out = super.transformRequest(model, body, stream, credentials);
|
||||
|
||||
// Preview models: thinking/params get defaults only — never override what the
|
||||
// caller set explicitly. (body.model is already `xiaomi/<id>` via upstreamModelId.)
|
||||
if (XiaomiMimoExecutor.isPreviewModel(model)) {
|
||||
if (out.thinking == null) out.thinking = { type: "enabled" };
|
||||
if (out.temperature == null) out.temperature = 1.0;
|
||||
if (out.top_p == null) out.top_p = 0.95;
|
||||
if (!out.max_tokens) out.max_tokens = 4096;
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
async execute(args) {
|
||||
const { model, credentials, proxyOptions = null } = args;
|
||||
if (!XiaomiMimoExecutor.isPreviewModel(model)) return super.execute(args);
|
||||
|
||||
const cookie = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions);
|
||||
if (!cookie) {
|
||||
throw new Error(
|
||||
"Xiaomi MiMo account session unavailable. Sign in to MiMo Desktop once so its passToken is present, then retry.",
|
||||
);
|
||||
}
|
||||
credentials[COOKIE_KEY] = cookie;
|
||||
const result = await super.execute(args);
|
||||
|
||||
// A cached session can expire early — drop it and retry once with a fresh one.
|
||||
if (result.response.status === 401) {
|
||||
invalidateMimoAccountCookieCache();
|
||||
const fresh = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions).catch(() => null);
|
||||
if (fresh) {
|
||||
credentials[COOKIE_KEY] = fresh;
|
||||
return super.execute(args);
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
export const __test__ = { PREVIEW_MODELS, bareModel, COOKIE_KEY };
|
||||
|
||||
export default XiaomiMimoExecutor;
|
||||
@@ -29,11 +29,18 @@ import {
|
||||
zedLlmFetch,
|
||||
} from "../shared/zedAuth.js";
|
||||
|
||||
// Wire values for the `provider` field of POST /completions. These are NOT
|
||||
// display names: cloud.zed.dev matches them exactly, and an unrecognized value
|
||||
// fails the whole request with `500 {"message":"An internal server error
|
||||
// occurred."}` before the model is ever looked at. Spellings come from Zed's
|
||||
// own GET /models catalog: `anthropic`, `open_ai`, `google` (note underscore),
|
||||
// `x_ai` follows the same convention — so feeding a catalog value back through
|
||||
// normalizeZedProvider is identity.
|
||||
const ZED_PROVIDER = {
|
||||
anthropic: "Anthropic",
|
||||
openai: "OpenAi",
|
||||
google: "Google",
|
||||
xai: "XAi",
|
||||
anthropic: "anthropic",
|
||||
openai: "open_ai",
|
||||
google: "google",
|
||||
xai: "x_ai",
|
||||
};
|
||||
|
||||
function normalizeZedProvider(value, model) {
|
||||
@@ -55,7 +62,14 @@ function buildProviderRequest(provider, model, body, stream, credentials) {
|
||||
return openaiToClaudeRequest(model, body, true);
|
||||
}
|
||||
if (provider === ZED_PROVIDER.google) {
|
||||
return openaiToGeminiRequest(model, body, true);
|
||||
const geminiRequest = openaiToGeminiRequest(model, body, true);
|
||||
// Zed's hosted Gemini backend speaks the Vertex safety vocabulary, not the
|
||||
// public Gemini API enum the shared translator emits (`OFF`, `CIVIC_INTEGRITY`,
|
||||
// `DANGEROUS_CONTENT`). Drop client-side safetySettings for the Zed Google
|
||||
// path so Zed applies its own defaults — scoped here so native Gemini/
|
||||
// Antigravity is untouched.
|
||||
delete geminiRequest.safetySettings;
|
||||
return geminiRequest;
|
||||
}
|
||||
if (provider === ZED_PROVIDER.openai) {
|
||||
return openaiToOpenAIResponsesRequest(model, body, true, credentials);
|
||||
|
||||
@@ -28,7 +28,7 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
|
||||
import { getCapabilitiesForModel } from "../providers/capabilities.js";
|
||||
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
|
||||
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
|
||||
import { defaultClaudeToolType } from "../translator/concerns/toolCall.js";
|
||||
import { defaultClaudeToolType, shouldDefaultClaudeToolType } from "../translator/concerns/toolCall.js";
|
||||
import { resolveSessionId } from "../utils/sessionManager.js";
|
||||
import { markPoolUnfit, clearPoolUnfit } from "../services/proxyPoolFitness.js";
|
||||
|
||||
@@ -249,7 +249,11 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
|
||||
// Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax)
|
||||
// reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing.
|
||||
if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) {
|
||||
// Provider-scoped via quirks (shouldDefaultClaudeToolType): only gateways that declare
|
||||
// requireClaudeToolType get the explicit type. Applying it unconditionally breaks
|
||||
// Claude-format endpoints that only accept the legacy typeless tool shape — DeepSeek's
|
||||
// Anthropic-compatible endpoint 400s with "unknown variant `custom`" (#3905).
|
||||
if (shouldDefaultClaudeToolType(provider, finalFormat, translatedBody.tools, PROVIDERS)) {
|
||||
translatedBody.tools = defaultClaudeToolType(translatedBody.tools);
|
||||
}
|
||||
|
||||
@@ -416,7 +420,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
const executeWithPoolFallback = async (attempt = 0) => {
|
||||
let result;
|
||||
try {
|
||||
result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
|
||||
result = await executor.execute({ model, body: translatedBody, stream, credentials, providerSessionId: sessionSeed, clientTool, signal: streamController.signal, log, proxyOptions });
|
||||
} catch (error) {
|
||||
if (error?.poolScoped && typeof resolveProxyConfig === "function" && attempt < MAX_POOL_RETRIES) {
|
||||
if (await tryNextPool(error.poolScoped, error.message)) return executeWithPoolFallback(attempt + 1);
|
||||
@@ -497,7 +501,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); }
|
||||
}
|
||||
try {
|
||||
const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
|
||||
const retryResult = await executor.execute({
|
||||
model,
|
||||
body: translatedBody,
|
||||
stream,
|
||||
credentials,
|
||||
providerSessionId: sessionSeed,
|
||||
clientTool,
|
||||
signal: streamController.signal,
|
||||
log,
|
||||
proxyOptions,
|
||||
});
|
||||
if (retryResult.response.ok) {
|
||||
providerResponse = retryResult.response;
|
||||
providerUrl = retryResult.url;
|
||||
@@ -556,7 +570,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
|
||||
|
||||
// Streaming response
|
||||
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId });
|
||||
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, credentials });
|
||||
}
|
||||
|
||||
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {
|
||||
|
||||
@@ -6,6 +6,7 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin
|
||||
import { createErrorResult } from "../../utils/error.js";
|
||||
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
|
||||
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
|
||||
import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js";
|
||||
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { decloakToolNames } from "../../utils/claudeCloaking.js";
|
||||
@@ -319,6 +320,11 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
|
||||
}
|
||||
}
|
||||
|
||||
// Unwrap before any consumer reads choices/usage so non-stream clients get a
|
||||
// bare OpenAI body and usage tracking sees data.usage. No-op unless the
|
||||
// provider opts in via transport.quirks.clineEnvelope.
|
||||
responseBody = unwrapClineEnvelope(responseBody, provider);
|
||||
|
||||
reqLogger.logProviderResponse(providerResponse.status, providerResponse.statusText, providerResponse.headers, responseBody);
|
||||
// Unwrap AFTER logging (raw envelope stays in the log for forensics) but
|
||||
// BEFORE usage extraction/translation so choices/usage resolve downstream.
|
||||
|
||||
@@ -3,8 +3,9 @@ import { needsTranslation } from "../../translator/index.js";
|
||||
import { createSSETransformStreamWithLogger, createPassthroughStreamWithLogger } from "../../utils/stream.js";
|
||||
import { pipeWithDisconnect } from "../../utils/streamHandler.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { HTTP_STATUS, STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
|
||||
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
|
||||
import { buildStreamErrorBytes } from "../../utils/streamHelpers.js";
|
||||
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
|
||||
import { saveRequestDetail } from "@/lib/usageDb.js";
|
||||
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
|
||||
@@ -22,7 +23,7 @@ const CODEX_SOURCE_TO_TARGET = {
|
||||
/**
|
||||
* Determine which SSE transform stream to use based on provider/format.
|
||||
*/
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) {
|
||||
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) {
|
||||
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
|
||||
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
|
||||
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
|
||||
@@ -30,11 +31,11 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
|
||||
if (needsCodexTranslation) {
|
||||
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
|
||||
}
|
||||
|
||||
if (needsTranslation(targetFormat, sourceFormat)) {
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
|
||||
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
|
||||
}
|
||||
|
||||
return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey);
|
||||
@@ -43,7 +44,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
|
||||
/**
|
||||
* Handle streaming response — pipe provider SSE through transform stream to client.
|
||||
*/
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) {
|
||||
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) {
|
||||
if (onRequestSuccess) {
|
||||
Promise.resolve()
|
||||
.then(onRequestSuccess)
|
||||
@@ -79,11 +80,16 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
|
||||
};
|
||||
}
|
||||
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey });
|
||||
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
|
||||
|
||||
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
|
||||
// Terminal bytes when the stream aborts after HTTP 200 was already sent, so the
|
||||
// client sees a real error instead of a silently truncated stream.
|
||||
// Responses passthrough keeps its own response.failed shape; every other client
|
||||
// format gets the OpenAI error frame + [DONE], or `event: error` for Claude.
|
||||
const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES;
|
||||
const onAbortTerminal = isResponsesPassthrough ? buildAbortedResponsesTerminalBytes : null;
|
||||
const onAbortTerminal = isResponsesPassthrough
|
||||
? buildAbortedResponsesTerminalBytes
|
||||
: (message) => buildStreamErrorBytes(HTTP_STATUS.GATEWAY_TIMEOUT, message, sourceFormat);
|
||||
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
|
||||
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs);
|
||||
|
||||
|
||||
@@ -2,13 +2,21 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { nowSec } from "./_base.js";
|
||||
import { PROVIDERS } from "../../config/providers.js";
|
||||
import { CODEX_CLI_VERSION } from "../../config/appConstants.js";
|
||||
|
||||
const CODEX_RESPONSES_URL = PROVIDERS["codex"].baseUrl;
|
||||
const CODEX_USER_AGENT = "codex_cli_rs/0.136.0";
|
||||
const CODEX_VERSION = "0.136.0";
|
||||
const CODEX_USER_AGENT = `codex_cli_rs/${CODEX_CLI_VERSION}`;
|
||||
const CODEX_ORIGINATOR = "codex_cli_rs";
|
||||
const CODEX_MODEL_SUFFIX = "-image";
|
||||
const CODEX_REF_DETAIL = "high";
|
||||
const CODEX_IMAGES_MAIN_MODEL = "gpt-5.5";
|
||||
const CODEX_TOOL_IMAGE_MODELS = new Set([
|
||||
"gpt-image-1.5",
|
||||
"gpt-image-2",
|
||||
"gpt-image-2.5",
|
||||
"gpt-image-2.5-flare",
|
||||
"gpt-image-2.5-sunburst",
|
||||
]);
|
||||
|
||||
function decodeAccountId(idToken) {
|
||||
try {
|
||||
@@ -27,6 +35,13 @@ function stripImageSuffix(model) {
|
||||
return model.endsWith(CODEX_MODEL_SUFFIX) ? model.slice(0, -CODEX_MODEL_SUFFIX.length) : model;
|
||||
}
|
||||
|
||||
function resolveCodexImageModels(model) {
|
||||
if (CODEX_TOOL_IMAGE_MODELS.has(model)) {
|
||||
return { responsesModel: CODEX_IMAGES_MAIN_MODEL, toolModel: model };
|
||||
}
|
||||
return { responsesModel: stripImageSuffix(model), toolModel: null };
|
||||
}
|
||||
|
||||
function toDataUrl(input) {
|
||||
if (!input || typeof input !== "string") return null;
|
||||
if (/^data:image\//i.test(input) || /^https?:\/\//i.test(input)) return input;
|
||||
@@ -157,7 +172,7 @@ export default {
|
||||
"originator": CODEX_ORIGINATOR,
|
||||
"session_id": randomUUID(),
|
||||
"user-agent": CODEX_USER_AGENT,
|
||||
"version": CODEX_VERSION,
|
||||
"version": CODEX_CLI_VERSION,
|
||||
"x-client-request-id": randomUUID(),
|
||||
};
|
||||
},
|
||||
@@ -167,21 +182,26 @@ export default {
|
||||
const single = toDataUrl(body.image);
|
||||
if (single) refs.push(single);
|
||||
const detail = body.image_detail || CODEX_REF_DETAIL;
|
||||
const { responsesModel, toolModel } = resolveCodexImageModels(model);
|
||||
const imgTool = { type: "image_generation", output_format: (body.output_format || "png").toLowerCase() };
|
||||
if (toolModel) {
|
||||
imgTool.action = refs.length > 0 ? "edit" : "generate";
|
||||
imgTool.model = toolModel;
|
||||
}
|
||||
if (body.size && body.size !== "") imgTool.size = body.size;
|
||||
if (body.quality && body.quality !== "") imgTool.quality = body.quality;
|
||||
if (body.background && body.background !== "") imgTool.background = body.background;
|
||||
return {
|
||||
model: stripImageSuffix(model),
|
||||
model: responsesModel,
|
||||
instructions: "",
|
||||
input: [{ type: "message", role: "user", content: buildContent(body.prompt, refs, detail) }],
|
||||
tools: [imgTool],
|
||||
tool_choice: "auto",
|
||||
tool_choice: toolModel ? { type: "image_generation" } : "auto",
|
||||
parallel_tool_calls: false,
|
||||
prompt_cache_key: randomUUID(),
|
||||
stream: true,
|
||||
store: false,
|
||||
reasoning: null,
|
||||
reasoning: toolModel ? { effort: "medium", summary: "auto" } : null,
|
||||
};
|
||||
},
|
||||
// Custom: codex parses SSE → either pipe to client or collect b64
|
||||
|
||||
@@ -2,6 +2,7 @@ import { createErrorResult } from "../utils/error.js";
|
||||
import { HTTP_STATUS } from "../config/runtimeConfig.js";
|
||||
import { refreshTokenByProvider } from "../services/tokenRefresh.js";
|
||||
import { PROVIDER_MEDIA } from "../providers/index.js";
|
||||
import { getVideoAdapter } from "./videoProviders/index.js";
|
||||
|
||||
// Upstream fetch deadline for video job submission/polling (the job itself is
|
||||
// async upstream — this only bounds the HTTP round-trip, not video rendering).
|
||||
@@ -94,21 +95,49 @@ export async function handleVideoProxyCore({
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Unknown video action: ${action}`);
|
||||
}
|
||||
|
||||
const method = requestId ? "GET" : "POST";
|
||||
const url = buildUpstreamUrl(config, action, requestId);
|
||||
const adapter = getVideoAdapter(provider);
|
||||
const fetchSignal = combineSignals(signal, timeoutMs);
|
||||
|
||||
const doFetch = (token) =>
|
||||
fetch(url, {
|
||||
// Default (xAI shape) request plan; adapters override URL/method/headers/body.
|
||||
const defaultPlan = () => {
|
||||
const method = requestId ? "GET" : "POST";
|
||||
return {
|
||||
method,
|
||||
headers: buildHeaders({ token, contentType: method === "POST" ? contentType : null, idempotencyKey: method === "POST" ? idempotencyKey : null }),
|
||||
url: buildUpstreamUrl(config, action, requestId),
|
||||
headers: buildHeaders({
|
||||
token: credentials?.accessToken || credentials?.apiKey,
|
||||
contentType: method === "POST" ? contentType : null,
|
||||
idempotencyKey: method === "POST" ? idempotencyKey : null,
|
||||
}),
|
||||
body: method === "POST" ? rawBody : undefined,
|
||||
signal: fetchSignal,
|
||||
});
|
||||
};
|
||||
};
|
||||
|
||||
// Rebuilt per attempt so the auth retry below picks up the refreshed token.
|
||||
const doFetch = async () => {
|
||||
const plan = adapter
|
||||
? await adapter.buildRequest({
|
||||
config, action, requestId, rawBody, contentType, idempotencyKey, credentials, log,
|
||||
token: credentials?.accessToken || credentials?.apiKey,
|
||||
})
|
||||
: defaultPlan();
|
||||
if (plan.error) return { planError: plan.error };
|
||||
return {
|
||||
response: await fetch(plan.url, {
|
||||
method: plan.method,
|
||||
headers: plan.headers,
|
||||
body: plan.body,
|
||||
signal: fetchSignal,
|
||||
}),
|
||||
};
|
||||
};
|
||||
|
||||
const method = requestId ? "GET" : "POST";
|
||||
let upstream;
|
||||
try {
|
||||
upstream = await doFetch(credentials?.accessToken || credentials?.apiKey);
|
||||
const first = await doFetch();
|
||||
if (first.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${first.planError}`);
|
||||
upstream = first.response;
|
||||
} catch (error) {
|
||||
if (error?.name === "AbortError" || error?.name === "TimeoutError") {
|
||||
return createErrorResult(HTTP_STATUS.REQUEST_TIMEOUT, `[${provider}] video ${method} aborted: ${error.message}`);
|
||||
@@ -136,7 +165,9 @@ export async function handleVideoProxyCore({
|
||||
await upstream.body?.cancel?.();
|
||||
} catch { /* noop */ }
|
||||
try {
|
||||
upstream = await doFetch(credentials.accessToken || credentials.apiKey);
|
||||
const retry = await doFetch();
|
||||
if (retry.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${retry.planError}`);
|
||||
upstream = retry.response;
|
||||
} catch (error) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video retry after refresh failed: ${error.message}`, credentials));
|
||||
}
|
||||
@@ -152,13 +183,25 @@ export async function handleVideoProxyCore({
|
||||
return createErrorResult(upstream.status, `[${provider}] ${message.slice(0, 2000)}`);
|
||||
}
|
||||
|
||||
// Success: pass the upstream JSON through untouched (request_id / status / video.url).
|
||||
// Success: pass the upstream JSON through untouched (request_id / status / video.url),
|
||||
// unless the adapter maps a provider-native shape onto it (Vertex operations).
|
||||
let outBody = bodyText;
|
||||
let outType = upstream.headers.get("content-type") || "application/json";
|
||||
if (adapter?.transformResponse) {
|
||||
try {
|
||||
outBody = JSON.stringify(adapter.transformResponse(JSON.parse(bodyText)));
|
||||
outType = "application/json";
|
||||
} catch {
|
||||
// Non-JSON or unexpected shape — fall back to the raw upstream body.
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
response: new Response(bodyText, {
|
||||
response: new Response(outBody, {
|
||||
status: upstream.status,
|
||||
headers: {
|
||||
"Content-Type": upstream.headers.get("content-type") || "application/json",
|
||||
"Content-Type": outType,
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
},
|
||||
}),
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
// Video provider adapters.
|
||||
//
|
||||
// Default (no adapter) = xAI shape: raw body forwarded to {baseUrl}/{action},
|
||||
// polled at {baseUrl}/{id}, upstream JSON passed through verbatim.
|
||||
// A provider only needs an adapter when its wire format differs from that.
|
||||
import openrouter from "./openrouter.js";
|
||||
import vertex from "./vertex.js";
|
||||
|
||||
const ADAPTERS = { openrouter, vertex };
|
||||
|
||||
export function getVideoAdapter(provider) {
|
||||
return ADAPTERS[provider] || null;
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
// OpenRouter video jobs — https://openrouter.ai/docs/api/api-reference/videos
|
||||
//
|
||||
// Same async shape as xAI (POST → { id, status }, GET → status/unsigned_urls),
|
||||
// two differences only: creation POSTs to the collection root (no `/generations`
|
||||
// suffix) and the account headers come from the registry entry.
|
||||
// Response bodies are passed through verbatim.
|
||||
|
||||
// ponytail: generations only — OpenRouter has no edits/extensions endpoint today.
|
||||
const SUPPORTED_ACTIONS = new Set(["generations"]);
|
||||
|
||||
function headers(config, token) {
|
||||
return {
|
||||
Accept: "application/json",
|
||||
...(config.headers || {}),
|
||||
...(token ? { Authorization: `Bearer ${token}` } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
export default {
|
||||
buildRequest({ config, action, requestId, rawBody, contentType, token }) {
|
||||
const base = config.baseUrl.replace(/\/$/, "");
|
||||
|
||||
if (requestId) {
|
||||
return { method: "GET", url: `${base}/${encodeURIComponent(requestId)}`, headers: headers(config, token) };
|
||||
}
|
||||
if (!SUPPORTED_ACTIONS.has(action)) {
|
||||
return { error: `OpenRouter video supports 'generations' only (got '${action}')` };
|
||||
}
|
||||
if (contentType && !contentType.includes("application/json")) {
|
||||
return { error: "OpenRouter video requires an application/json body" };
|
||||
}
|
||||
return {
|
||||
method: "POST",
|
||||
url: base,
|
||||
headers: { ...headers(config, token), "Content-Type": "application/json" },
|
||||
body: rawBody,
|
||||
};
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,159 @@
|
||||
// Vertex AI (Veo) video jobs.
|
||||
//
|
||||
// Vertex does NOT speak the OpenAI-ish /v1/videos shape, so unlike OpenRouter
|
||||
// this adapter translates both directions:
|
||||
// create → POST {model}:predictLongRunning { instances[], parameters{} } → { name }
|
||||
// poll → POST {model}:fetchPredictOperation { operationName } → { done, response }
|
||||
// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation
|
||||
//
|
||||
// The operation name is a resource path (contains "/"), so it is base64url-encoded
|
||||
// into the job id returned to the client — GET /v1/videos/{id} stays a flat path.
|
||||
import { parseVertexSaJson, refreshVertexToken } from "../../services/tokenRefresh.js";
|
||||
|
||||
const DEFAULT_LOCATION = "us-central1";
|
||||
|
||||
const encodeJobId = (name) => Buffer.from(name, "utf8").toString("base64url");
|
||||
|
||||
// Operation name shape: projects/{p}/locations/{l}/publishers/{pub}/models/{m}/operations/{op}.
|
||||
// Anchored and single-segment-per-field so a decoded path can never carry `..` or a
|
||||
// host-changing prefix into the request URL.
|
||||
const OPERATION_NAME_RE = /^projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/models\/[^/]+\/operations\/[^/]+$/;
|
||||
|
||||
function modelPathOf(operationName) {
|
||||
return operationName.slice(0, operationName.indexOf("/operations/"));
|
||||
}
|
||||
|
||||
function decodeJobId(id) {
|
||||
const raw = String(id ?? "");
|
||||
// Buffer.from(x, "base64url") silently drops invalid characters instead of
|
||||
// throwing, so only ids that re-encode byte-for-byte are accepted.
|
||||
if (!raw || raw.length > 1024 || !/^[A-Za-z0-9_-]+$/.test(raw)) return null;
|
||||
const decoded = Buffer.from(raw, "base64url").toString("utf8");
|
||||
if (Buffer.from(decoded, "utf8").toString("base64url") !== raw) return null;
|
||||
return OPERATION_NAME_RE.test(decoded) ? decoded : null;
|
||||
}
|
||||
|
||||
async function resolveAuth(credentials, log) {
|
||||
const saJson = parseVertexSaJson(credentials?.apiKey);
|
||||
const projectId =
|
||||
saJson?.project_id ||
|
||||
credentials?.projectId ||
|
||||
credentials?.providerSpecificData?.projectId;
|
||||
const location = credentials?.providerSpecificData?.location || DEFAULT_LOCATION;
|
||||
|
||||
if (!projectId) {
|
||||
return { error: "Vertex video requires a project_id — use Service Account JSON or set providerSpecificData.projectId" };
|
||||
}
|
||||
|
||||
let token = credentials?.accessToken;
|
||||
if (saJson) {
|
||||
const minted = await refreshVertexToken(saJson, log);
|
||||
if (!minted?.accessToken) return { error: "Vertex video: failed to mint access token from service account JSON" };
|
||||
token = minted.accessToken;
|
||||
}
|
||||
if (!token) return { error: "Vertex video requires Service Account JSON or an OAuth access token (raw API keys are not supported)" };
|
||||
|
||||
return { token, projectId, location };
|
||||
}
|
||||
|
||||
/** OpenAI-ish video body → Vertex predictLongRunning body. */
|
||||
function toVertexBody(body) {
|
||||
const instance = { prompt: body.prompt };
|
||||
// Image-to-video: accept the Vertex-native shape or a bare data URL / base64 string.
|
||||
const image = body.image ?? body.image_url;
|
||||
if (image && typeof image === "object") {
|
||||
instance.image = image;
|
||||
} else if (typeof image === "string") {
|
||||
const match = image.match(/^data:([^;]+);base64,(.*)$/s);
|
||||
instance.image = match
|
||||
? { bytesBase64Encoded: match[2], mimeType: match[1] }
|
||||
: { gcsUri: image };
|
||||
}
|
||||
if (body.video && typeof body.video === "object") instance.video = body.video;
|
||||
|
||||
const parameters = {};
|
||||
if (body.n != null) parameters.sampleCount = Number(body.n);
|
||||
if (body.duration != null) parameters.durationSeconds = Number(body.duration);
|
||||
if (body.aspect_ratio) parameters.aspectRatio = body.aspect_ratio;
|
||||
if (body.resolution) parameters.resolution = body.resolution;
|
||||
if (body.seed != null) parameters.seed = body.seed;
|
||||
if (body.negative_prompt) parameters.negativePrompt = body.negative_prompt;
|
||||
// Without storageUri Vertex returns inline base64 bytes; a GCS bucket keeps
|
||||
// the poll response small and is what production callers want.
|
||||
if (body.storage_uri) parameters.storageUri = body.storage_uri;
|
||||
if (body.generate_audio != null) parameters.generateAudio = !!body.generate_audio;
|
||||
|
||||
return { instances: [instance], ...(Object.keys(parameters).length ? { parameters } : {}) };
|
||||
}
|
||||
|
||||
/** Vertex operation → the async-job shape 9Router clients already poll for. */
|
||||
function fromVertexOperation(json) {
|
||||
if (!json?.name) return json;
|
||||
const id = encodeJobId(json.name);
|
||||
if (json.error) {
|
||||
return { id, request_id: id, status: "failed", error: json.error };
|
||||
}
|
||||
if (!json.done) {
|
||||
return { id, request_id: id, status: "pending" };
|
||||
}
|
||||
const samples =
|
||||
json.response?.videos ||
|
||||
json.response?.generateVideoResponse?.generatedSamples ||
|
||||
[];
|
||||
const videos = samples.map((s) => ({
|
||||
url: s.gcsUri || s.video?.uri || s.uri || null,
|
||||
b64_json: s.bytesBase64Encoded || s.video?.bytesBase64Encoded || null,
|
||||
mime_type: s.mimeType || s.video?.mimeType || "video/mp4",
|
||||
}));
|
||||
return { id, request_id: id, status: "completed", video: videos[0] || null, videos };
|
||||
}
|
||||
|
||||
export default {
|
||||
async buildRequest({ config, action, requestId, rawBody, contentType, credentials, log }) {
|
||||
if (contentType && !contentType.includes("application/json")) {
|
||||
return { error: "Vertex video requires an application/json body" };
|
||||
}
|
||||
|
||||
const auth = await resolveAuth(credentials, log);
|
||||
if (auth.error) return { error: auth.error };
|
||||
const { token, projectId, location } = auth;
|
||||
const base = (config.baseUrl || "https://aiplatform.googleapis.com").replace(/\/$/, "");
|
||||
const headers = { Accept: "application/json", "Content-Type": "application/json", Authorization: `Bearer ${token}` };
|
||||
|
||||
if (requestId) {
|
||||
const operationName = decodeJobId(requestId);
|
||||
if (!operationName) return { error: "Invalid Vertex video job id" };
|
||||
return {
|
||||
method: "POST",
|
||||
url: `${base}/v1/${modelPathOf(operationName)}:fetchPredictOperation`,
|
||||
headers,
|
||||
body: JSON.stringify({ operationName }),
|
||||
};
|
||||
}
|
||||
|
||||
if (action !== "generations") {
|
||||
// ponytail: Veo extend/edit go through generations with `video`/`image` in the body.
|
||||
return { error: `Vertex video supports 'generations' only (got '${action}')` };
|
||||
}
|
||||
|
||||
let body;
|
||||
try {
|
||||
body = JSON.parse(typeof rawBody === "string" ? rawBody : rawBody.toString("utf8"));
|
||||
} catch {
|
||||
return { error: "Invalid JSON body" };
|
||||
}
|
||||
if (!body.model) return { error: "Vertex video requires a model (e.g. vertex/veo-3.1-generate-preview)" };
|
||||
// Plain model id only — a path segment carrying "/" or ".." would rewrite the URL.
|
||||
if (!/^[A-Za-z0-9._-]+$/.test(body.model)) return { error: "Invalid Vertex video model id" };
|
||||
if (!body.prompt && !body.image && !body.image_url) return { error: "Vertex video requires a prompt or an image" };
|
||||
|
||||
return {
|
||||
method: "POST",
|
||||
url: `${base}/v1/projects/${projectId}/locations/${location}/publishers/google/models/${body.model}:predictLongRunning`,
|
||||
headers,
|
||||
body: JSON.stringify(toVertexBody(body)),
|
||||
};
|
||||
},
|
||||
|
||||
transformResponse: fromVertexOperation,
|
||||
};
|
||||
@@ -116,6 +116,16 @@ export const MODEL_CAPABILITIES = {
|
||||
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
|
||||
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
|
||||
// DeepSeek V4.1-Flash is natively multimodal — models.dev lists
|
||||
// opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream
|
||||
// the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the
|
||||
// same image capability as the exp id above. "deepseek-flash" is the GA id on the
|
||||
// DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose
|
||||
// 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry
|
||||
// short-circuits the pattern table, so a vision-only delta would drop them.
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
"deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 },
|
||||
|
||||
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
|
||||
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
|
||||
"coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
|
||||
@@ -131,6 +141,8 @@ export const MODEL_CAPABILITIES = {
|
||||
// via OpenAI Responses input_image; reasoning supports up to xhigh.
|
||||
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
|
||||
// OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output
|
||||
"union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 },
|
||||
};
|
||||
|
||||
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
|
||||
@@ -154,6 +166,7 @@ export const PROVIDER_CAPABILITIES = {
|
||||
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
|
||||
},
|
||||
"codex": {
|
||||
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
|
||||
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
|
||||
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
|
||||
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
|
||||
@@ -178,39 +191,95 @@ export const PROVIDER_CAPABILITIES = {
|
||||
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
|
||||
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
|
||||
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
|
||||
// (see registry thinkingFormat). `onlyReasoning` models can't turn thinking
|
||||
// off → thinkingCanDisable:false (clamped to minimal instead of disabled).
|
||||
// (see registry thinkingFormat). For thinkingCanDisable use the server's
|
||||
// reasoning.canDisableThinking flag — see the note in the codebuddy-cn block
|
||||
// below; it is NOT the inverse of onlyReasoning.
|
||||
"codebuddy-cn": {
|
||||
"glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 },
|
||||
// maxOutput 64000 per both the plugin-baked fallback and the live server
|
||||
// table (the old 38000 had no source and truncated output).
|
||||
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 },
|
||||
"glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
|
||||
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 },
|
||||
"minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
|
||||
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
|
||||
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
|
||||
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 },
|
||||
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
|
||||
// hy3/hy3-x: 256K official (192K conservative, matches hy3-preview); hy4-preview: 1M official.
|
||||
// glm-5.3: 1M (GLM-5.x gen); glm-5.3-flash window unverified (200K conservative).
|
||||
// Per-model values mirror the server's product-config payload (the plugin
|
||||
// fetches it from copilot.tencent.com; the `models[]` entries carry
|
||||
// maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
|
||||
// maxInputTokens, maxOutput = maxOutputTokens. Where the server and the
|
||||
// plugin-baked fallback disagree, the server table wins.
|
||||
// ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking —
|
||||
// it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is
|
||||
// on by default"; canDisableThinking means "it CAN be turned off". glm-5.3
|
||||
// and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so
|
||||
// their thinking is switchable; the hy* models are forced always-on.
|
||||
"hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
|
||||
"hy3-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
|
||||
"hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
|
||||
"hy4-preview-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
|
||||
"glm-5.3": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.3-flash": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
|
||||
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
|
||||
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
|
||||
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 },
|
||||
"glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
|
||||
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
|
||||
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
|
||||
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
|
||||
// deepseek-v4.1-flash replaces v4-flash (dropped from the server list;
|
||||
// the old endpoint still answers 200 but the published list is the
|
||||
// contract). maxOutput 128000 per the server's product-config payload.
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
|
||||
},
|
||||
// CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors
|
||||
// the codebuddy-cn entry (the openai-style reasoning_effort format matters:
|
||||
// the generic *deepseek-v4* pattern would otherwise pick the vendor-native
|
||||
// "deepseek" thinking shape, which the CodeBuddy gateway does not accept).
|
||||
"codebuddy-intl": {
|
||||
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
|
||||
},
|
||||
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
|
||||
// registry `name` is display-only and capability lookup matches on the raw
|
||||
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
|
||||
// (200K) without this map. contextWindow follows the real model family's
|
||||
// spec: the /algo/api/v2/model/list max_input_tokens under-reports some
|
||||
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
|
||||
// max_output_tokens arrives as 0 for every model, so outputs are
|
||||
// best-guess from the real model family. Vision tags below follow the
|
||||
// upstream is_vl flag. The executor uploads inlined images to
|
||||
// /api/v2/image/upload and leaves image_urls/chat_context.imageUrls null
|
||||
// (same as qodercli). reasoning:true on all of them — every model can
|
||||
// reason; the upstream is_reasoning flag only drives model_config selection.
|
||||
// thinkingFormat keeps the true-model family for documentation/UI, but
|
||||
// thinkingCanDisable:false everywhere: the executor only forwards
|
||||
// messages/tools/max_tokens, and thinking is fixed upstream via
|
||||
// modelConfig.is_reasoning — client thinking intent is dropped, so "none"
|
||||
// must never be offered as an option.
|
||||
"qoder": {
|
||||
"ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5
|
||||
"performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5
|
||||
"dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro
|
||||
"dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash
|
||||
"gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3
|
||||
"gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash
|
||||
"kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3
|
||||
"kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code
|
||||
"mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3
|
||||
"qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max
|
||||
"qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus
|
||||
"qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash
|
||||
"qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max
|
||||
},
|
||||
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
|
||||
"poolside": {
|
||||
"laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 },
|
||||
"laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 },
|
||||
},
|
||||
// Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge
|
||||
// the library page publishes for this model (text+image in, 1M context).
|
||||
// ponytail: thinkingFormat stays "deepseek" to preserve today's body shape;
|
||||
// Ollama's native toggle is the top-level `think` field (bool or
|
||||
// low/medium/high/max), which no format in thinkingUnified.js emits yet —
|
||||
// openai-to-ollama.js drops it. Wire a "think" format when thinking on
|
||||
// Ollama Cloud is actually needed.
|
||||
"ollama": {
|
||||
"deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -247,6 +316,9 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } },
|
||||
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
|
||||
|
||||
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
|
||||
|
||||
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
|
||||
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
|
||||
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
|
||||
@@ -312,7 +384,11 @@ export const PATTERN_CAPABILITIES = [
|
||||
{ pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
|
||||
|
||||
// ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─
|
||||
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } },
|
||||
// v4.1+ has real image input (probed live on Alibaba MaaS: correct color
|
||||
// read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore
|
||||
// them (answered "Unknown"), so vision stays scoped to v4.* dotted releases.
|
||||
{ pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } },
|
||||
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } },
|
||||
{ pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
|
||||
{ pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
|
||||
{ pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } },
|
||||
@@ -377,14 +453,27 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
|
||||
|
||||
// Catalog lookups, installed by the server at startup. Left as no-ops in the
|
||||
// browser bundle, where there is no file to read.
|
||||
//
|
||||
// The server bundles this module into every route chunk that needs it, and each
|
||||
// copy carries its own module state, so an install landing in the copy the
|
||||
// startup hook imported stays invisible to the copy resolving requests. The slot
|
||||
// lives on globalThis instead; the local binding is the fast path.
|
||||
let catalogSource = null;
|
||||
|
||||
/**
|
||||
* Install the synced catalog reader (server only).
|
||||
* @param {{ getModalities: Function, getLimits: Function } | null} source
|
||||
* @param {{ getModalities: (provider: string, model: string) => object|null,
|
||||
* getLimits: (provider: string, model: string) => object|null } | null} source
|
||||
*/
|
||||
export function setCatalogSource(source) {
|
||||
catalogSource = source;
|
||||
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source;
|
||||
}
|
||||
|
||||
function getCatalogSource() {
|
||||
if (catalogSource) return catalogSource;
|
||||
if (typeof globalThis === "undefined") return null;
|
||||
return (catalogSource = globalThis.__9rCatalogSource || null);
|
||||
}
|
||||
|
||||
// Apply the synced catalog + name heuristic on top of a table-resolved result.
|
||||
@@ -393,15 +482,16 @@ export function setCatalogSource(source) {
|
||||
function refine(base, provider, model) {
|
||||
const result = { ...DEFAULT_CAPABILITIES, ...base };
|
||||
|
||||
if (catalogSource) {
|
||||
const modalities = catalogSource.getModalities(model);
|
||||
const source = getCatalogSource();
|
||||
if (source) {
|
||||
const modalities = source.getModalities(provider, model);
|
||||
if (modalities) {
|
||||
for (const key of MODALITY_KEYS) {
|
||||
if (modalities[key] === true) result[key] = true;
|
||||
}
|
||||
}
|
||||
|
||||
const limits = catalogSource.getLimits(provider, model);
|
||||
const limits = source.getLimits(provider, model);
|
||||
if (limits) {
|
||||
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
|
||||
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
|
||||
@@ -413,12 +503,66 @@ function refine(base, provider, model) {
|
||||
return result;
|
||||
}
|
||||
|
||||
// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models
|
||||
// default to vision; only this denylist stays text-only.
|
||||
const COMMANDCODE_TEXT_ONLY = new Set([
|
||||
"deepseek/deepseek-v4-pro",
|
||||
"deepseek/deepseek-v4-flash",
|
||||
"deepseek/deepseek-v4-flash-fast",
|
||||
"zai-org/glm-5.3",
|
||||
"zai-org/glm-5.2",
|
||||
"zai-org/glm-5.2-fast",
|
||||
"zai-org/glm-5.1",
|
||||
"zai-org/glm-5",
|
||||
"minimaxai/minimax-m2.7",
|
||||
"minimax/minimax-m2.7-free",
|
||||
"minimaxai/minimax-m2.5",
|
||||
"xiaomi/mimo-v2.5-pro",
|
||||
"qwen/qwen3.6-max-preview",
|
||||
"qwen/qwen3.7-max",
|
||||
"meituan/longcat-2.0:free",
|
||||
"stepfun/step-3.5-flash",
|
||||
"tencent/hy4-preview",
|
||||
"tencent/hy3",
|
||||
"tencent/hy3-paid",
|
||||
"nvidia/nemotron-3-ultra-550b-a55b",
|
||||
"poolside/laguna-s-2.1-free",
|
||||
"inclusionai/ling-3.0-flash-free",
|
||||
"inclusionai/ling-3.0-flash-sante:free",
|
||||
]);
|
||||
|
||||
function isCommandCodeTextOnly(model) {
|
||||
const key = String(model || "").toLowerCase();
|
||||
if (COMMANDCODE_TEXT_ONLY.has(key)) return true;
|
||||
for (const id of COMMANDCODE_TEXT_ONLY) {
|
||||
const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
|
||||
if (key === base || key.endsWith("/" + base)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
export function getCapabilitiesForModel(provider, model) {
|
||||
if (!model) return { ...DEFAULT_CAPABILITIES };
|
||||
|
||||
// Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7".
|
||||
const baseModel = model.includes("/") ? model.split("/").pop() : model;
|
||||
|
||||
// CommandCode wire is /alpha/generate for every model. Family patterns
|
||||
// (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here.
|
||||
if (provider === "commandcode" || provider === "cmc") {
|
||||
const providerCaps = PROVIDER_CAPABILITIES.commandcode;
|
||||
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
|
||||
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
|
||||
return {
|
||||
...DEFAULT_CAPABILITIES,
|
||||
reasoning: true,
|
||||
thinkingFormat: "commandcode",
|
||||
thinkingEffortSupported: true,
|
||||
vision: !isCommandCodeTextOnly(model),
|
||||
contextWindow: 1000000,
|
||||
maxOutput: 384000,
|
||||
};
|
||||
}
|
||||
|
||||
// 1. Provider-specific override
|
||||
if (provider) {
|
||||
const providerCaps = PROVIDER_CAPABILITIES[provider];
|
||||
|
||||
@@ -13,6 +13,11 @@ export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
|
||||
// Trimmed upstream catalog, read by the add-models skill (not by the router).
|
||||
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
|
||||
|
||||
// Schema of the file this module reads. The writer stamps it; a file carrying an
|
||||
// older value predates provider-scoped modality keys, and its flat keys are not
|
||||
// looked up here, so the sync rebuilds it instead of asking upstream for a 304.
|
||||
export const CATALOG_VERSION = 2;
|
||||
|
||||
const EMPTY = { models: {}, providers: {} };
|
||||
let cache = EMPTY;
|
||||
let cachedMtime = -1;
|
||||
@@ -45,14 +50,19 @@ function load() {
|
||||
return cache;
|
||||
}
|
||||
|
||||
// Modality is a property of the model itself — any gateway serving it inherits
|
||||
// the same image/video/pdf support, so this is keyed by model id alone.
|
||||
export function getCatalogModalities(model) {
|
||||
return load().models[baseId(model)] || null;
|
||||
// Modalities are recorded per gateway upstream, and gateways disagree about the
|
||||
// same weights — some do not proxy images at all — so the key is provider +
|
||||
// model, in the local provider id space, exactly like the limits below. Keying
|
||||
// by model id alone made short ids collide across vendors: "auto", "free" and
|
||||
// "efficient" are router modes in one catalog and model names in another, and a
|
||||
// request to the router mode inherited a stranger's vision.
|
||||
export function getCatalogModalities(provider, model) {
|
||||
if (!provider) return null;
|
||||
return load().models[`${provider}:${baseId(model)}`] || null;
|
||||
}
|
||||
|
||||
// Context and output limits are a property of the gateway, not the model: each
|
||||
// one truncates differently, so these stay keyed by provider + model.
|
||||
// Context and output limits are a property of the gateway too: each one
|
||||
// truncates differently, so these stay keyed by provider + model.
|
||||
export function getCatalogLimits(provider, model) {
|
||||
const byProvider = provider && load().providers[provider];
|
||||
if (!byProvider) return null;
|
||||
|
||||
@@ -53,6 +53,7 @@ export const MODEL_PRICING = {
|
||||
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
|
||||
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
|
||||
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
|
||||
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
|
||||
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
|
||||
|
||||
@@ -110,6 +111,8 @@ export const MODEL_PRICING = {
|
||||
"deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
|
||||
"deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 },
|
||||
|
||||
// === GLM ===
|
||||
|
||||
@@ -37,6 +37,7 @@ export default {
|
||||
},
|
||||
usage: {
|
||||
quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
|
||||
quotaSummaryApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:retrieveUserQuotaSummary`,
|
||||
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
|
||||
tokenUrl: "https://oauth2.googleapis.com/token",
|
||||
},
|
||||
|
||||
@@ -20,6 +20,8 @@ export default {
|
||||
authModes: [
|
||||
"apikey",
|
||||
],
|
||||
passthroughModels: true,
|
||||
modelsFetcher: { url: "https://api.airforce/v1/models", type: "airforce-free" },
|
||||
transport: {
|
||||
baseUrl: "https://api.airforce/v1/chat/completions",
|
||||
validateUrl: "https://api.airforce/v1/models",
|
||||
@@ -27,10 +29,11 @@ export default {
|
||||
"HTTP-Referer": "https://endpoint-proxy.local",
|
||||
"X-Title": "Endpoint Proxy",
|
||||
},
|
||||
forceStream: true,
|
||||
},
|
||||
models: [
|
||||
{ id: "anthropic/claude-3.7-sonnet", name: "Claude 3.7 Sonnet (Free)", contextLength: 200000 },
|
||||
{ id: "moonshot/kimi-k2.6", name: "Kimi K2.6 (Free)", contextLength: 262144 },
|
||||
{ id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash (Free)", contextLength: 1048576 },
|
||||
{ id: "gpt-oss-120b", name: "GPT-OSS 120B (Free)", contextLength: 131072 },
|
||||
{ id: "gpt-oss-20b", name: "GPT-OSS 20B (Free)", contextLength: 131072 },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code (Free)", contextLength: 262144 },
|
||||
],
|
||||
};
|
||||
|
||||
@@ -23,6 +23,8 @@ export default {
|
||||
"HTTP-Referer": "https://cline.bot",
|
||||
"X-Title": "Cline",
|
||||
},
|
||||
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
|
||||
quirks: { clineEnvelope: true },
|
||||
tokenUrl: "https://api.cline.bot/api/v1/auth/token",
|
||||
refreshUrl: "https://api.cline.bot/api/v1/auth/refresh",
|
||||
auth: {
|
||||
|
||||
@@ -14,7 +14,10 @@ export default {
|
||||
},
|
||||
},
|
||||
category: "oauth",
|
||||
authModes: ["oauth", "apikey"],
|
||||
// ClinePass authenticates with a plain API key from app.cline.bot/settings/api-keys
|
||||
// (category "apikey"). The OAuth extension flow used by Cline does not issue
|
||||
// tokens that the ClinePass API consumer endpoint accepts (HTTP 401) — see #2333.
|
||||
authModes: ["apikey", "oauth"],
|
||||
hasOAuth: true,
|
||||
transport: {
|
||||
baseUrl: "https://api.cline.bot/api/v1/chat/completions",
|
||||
@@ -22,6 +25,8 @@ export default {
|
||||
"HTTP-Referer": "https://cline.bot",
|
||||
"X-Title": "Cline",
|
||||
},
|
||||
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
|
||||
quirks: { clineEnvelope: true },
|
||||
auth: {
|
||||
combined: true,
|
||||
header: "Authorization",
|
||||
|
||||
@@ -47,27 +47,28 @@ export default {
|
||||
models: [
|
||||
{ id: "glm-5.2", name: "GLM-5.2" },
|
||||
{ id: "glm-5.1", name: "GLM-5.1" },
|
||||
{ id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" },
|
||||
{ id: "glm-5v-turbo", name: "GLM-5v-Turbo" },
|
||||
{ id: "minimax-m3", name: "MiniMax-M3" },
|
||||
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
|
||||
{ id: "kimi-k2.7", name: "Kimi-K2.7-Code" },
|
||||
{ id: "kimi-k2.6", name: "Kimi-K2.6" },
|
||||
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
|
||||
// "-x" suffix = paid tier of the same model (free id rides the promo quota:
|
||||
// hy3 free until 2026-08-31, hy4-preview until 2026-09-10). Server model table
|
||||
// seen in client logs 2026-08-30; glm-5.0 / glm-4.7 removed (API 11102 dead).
|
||||
{ id: "hy3-preview", name: "Hy3 Preview" },
|
||||
// Catalog mirrors the server's product-config payload (the plugin fetches
|
||||
// it from copilot.tencent.com). Models the server no longer publishes are
|
||||
// removed even when the chat endpoint still answers them — the published
|
||||
// list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x
|
||||
// (endpoint returns 11102 "model service info not found"), plus
|
||||
// glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview /
|
||||
// deepseek-v3-2-volc (absent from the server list, though still answering
|
||||
// 200) and hy3-x (paid tier, not used here). deepseek-v4-flash removed
|
||||
// 2026-09: replaced server-side by deepseek-v4.1-flash (same low/high/
|
||||
// xhigh efforts; endpoint still answers 200 but the list is the contract).
|
||||
// "-x" suffix = paid tier of the same model (free id rides the promo quota).
|
||||
{ id: "hy3", name: "Hy3" },
|
||||
{ id: "hy3-x", name: "Hy3 (Paid)" },
|
||||
{ id: "hy4-preview", name: "Hy4-Preview" },
|
||||
{ id: "hy4-preview-x", name: "Hy4-Preview (Paid)" },
|
||||
{ id: "glm-5.3", name: "GLM-5.3" },
|
||||
{ id: "glm-5.3-flash", name: "GLM-5.3-Flash" },
|
||||
{ id: "kimi-k3-1", name: "Kimi-K3" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
|
||||
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
|
||||
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
|
||||
],
|
||||
oauth: {
|
||||
baseUrl: "https://copilot.tencent.com",
|
||||
|
||||
@@ -58,7 +58,9 @@ export default {
|
||||
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
|
||||
{ id: "hy3-preview", name: "Hy3 Preview" },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
|
||||
// deepseek-v4-flash replaced server-side by deepseek-v4.1-flash (same
|
||||
// catalog as CN; the old endpoint still answers 200 but the list is the contract).
|
||||
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
|
||||
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
|
||||
],
|
||||
oauth: {
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
import { withCodexReviewModels } from "../models/helpers.js";
|
||||
|
||||
// Codex CLI version seen by OpenAI's backend — single source for the Version /
|
||||
// User-Agent identity headers. Bump when the installed codex CLI is upgraded.
|
||||
const CODEX_CLI_VERSION = "0.154.0";
|
||||
|
||||
export default {
|
||||
id: "codex",
|
||||
priority: 30,
|
||||
@@ -34,9 +38,10 @@ export default {
|
||||
baseUrl: "https://chatgpt.com/backend-api/codex/responses",
|
||||
format: "openai-responses",
|
||||
forceStream: true,
|
||||
cliVersion: CODEX_CLI_VERSION,
|
||||
headers: {
|
||||
originator: "codex_cli_rs",
|
||||
"User-Agent": "codex_cli_rs/0.136.0",
|
||||
"User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`,
|
||||
},
|
||||
usage: {
|
||||
url: "https://chatgpt.com/backend-api/wham/usage",
|
||||
@@ -45,6 +50,7 @@ export default {
|
||||
},
|
||||
},
|
||||
models: [
|
||||
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
|
||||
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
|
||||
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
|
||||
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
|
||||
@@ -59,6 +65,17 @@ export default {
|
||||
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
|
||||
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
|
||||
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
|
||||
// Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived
|
||||
// from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398).
|
||||
{ id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" },
|
||||
{ id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-image-2", name: "GPT Image 2", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-image-1.5", name: "GPT Image 1.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
{ id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
|
||||
|
||||
@@ -40,4 +40,8 @@ export default {
|
||||
{ id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" },
|
||||
{ id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" },
|
||||
],
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -25,6 +25,21 @@ export default {
|
||||
reasoningInject: {
|
||||
scope: "all",
|
||||
},
|
||||
quirks: {
|
||||
// DeepSeek's Anthropic-compatible endpoint
|
||||
// (https://api.deepseek.com/anthropic/v1/messages) accepts ONLY the
|
||||
// built-in web_search_* tools and rejects client-defined `custom` tools
|
||||
// (MCP / Read / Bash / etc.) with HTTP 400
|
||||
// "tools[0]: unknown variant `custom`, expected
|
||||
// `web_search_20250305` or `web_search_20260209`".
|
||||
//
|
||||
// Declaring this whitelist makes prepareClaudeRequest() forward only
|
||||
// web_search_* tools and strip everything else before sending, so MCP /
|
||||
// function tools are dropped instead of failing the whole request.
|
||||
// DeepSeek's OpenAI-compatible transport is unaffected (targetFormat
|
||||
// there is "openai", not "claude", so prepareClaudeRequest is not run).
|
||||
claudeSupportedToolTypes: ["web_search_20250305", "web_search_20260209"],
|
||||
},
|
||||
},
|
||||
// Multi-endpoint: pick the transport matching client sourceFormat to skip translation.
|
||||
transports: [
|
||||
@@ -44,6 +59,7 @@ export default {
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
|
||||
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
|
||||
{ id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash" },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
|
||||
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
|
||||
|
||||
@@ -25,6 +25,7 @@ export default {
|
||||
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
{ id: "glm-4.7", name: "GLM-4.7" },
|
||||
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
|
||||
|
||||
@@ -49,6 +49,7 @@ export default {
|
||||
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
|
||||
{ id: "glm-5.2", name: "GLM 5.2" },
|
||||
{ id: "glm-5.1", name: "GLM 5.1" },
|
||||
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
|
||||
{ id: "glm-5", name: "GLM 5" },
|
||||
{ id: "glm-4.7", name: "GLM 4.7" },
|
||||
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
|
||||
|
||||
@@ -22,6 +22,7 @@ export default {
|
||||
headers: { ...CLAUDE_API_HEADERS },
|
||||
quirks: {
|
||||
dropOutputConfig: true,
|
||||
requireClaudeToolType: true,
|
||||
},
|
||||
reasoningInject: {
|
||||
scope: "all",
|
||||
|
||||
@@ -22,6 +22,7 @@ export default {
|
||||
headers: { ...CLAUDE_API_HEADERS },
|
||||
quirks: {
|
||||
dropOutputConfig: true,
|
||||
requireClaudeToolType: true,
|
||||
},
|
||||
reasoningInject: {
|
||||
scope: "all",
|
||||
|
||||
@@ -30,6 +30,7 @@ export default {
|
||||
{ id: "glm-4.7-flash", name: "GLM 4.7 Flash" },
|
||||
{ id: "qwen3.5", name: "Qwen3.5" },
|
||||
{ id: "minimax-m3", name: "MiniMax M3" },
|
||||
{ id: "deepseek-v4.1-flash:cloud", name: "DeepSeek V4.1 Flash" },
|
||||
],
|
||||
serviceKinds: ["llm", "webFetch"],
|
||||
fetchConfig: {
|
||||
|
||||
@@ -57,6 +57,9 @@ export default {
|
||||
{ id: "whisper-1", name: "Whisper 1", params: ["language","response_format","temperature","prompt"], kind: "stt" },
|
||||
{ id: "gpt-4o-transcribe", name: "GPT-4o Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
|
||||
{ id: "gpt-4o-mini-transcribe", name: "GPT-4o Mini Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
|
||||
{ id: "gpt-image-2.5", name: "GPT Image 2.5", params: ["n","size","quality","response_format"], kind: "image" },
|
||||
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", params: ["n","size","quality","response_format"], kind: "image" },
|
||||
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", params: ["n","size","quality","response_format"], kind: "image" },
|
||||
{ id: "gpt-image-1", name: "GPT Image 1", params: ["n","size","quality","response_format"], kind: "image" },
|
||||
{ id: "dall-e-3", name: "DALL-E 3", params: ["size","quality","style","response_format"], kind: "image" },
|
||||
{ id: "dall-e-2", name: "DALL-E 2", params: ["n","size","response_format"], kind: "image" },
|
||||
|
||||
@@ -13,7 +13,7 @@ export default {
|
||||
textIcon: "OC",
|
||||
website: "https://opencode.ai/auth",
|
||||
notice: {
|
||||
text: "OpenCode Go subscription: $5/mo (then 0/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
|
||||
text: "OpenCode Go subscription: $5/mo (then 10/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
|
||||
apiKeyUrl: "https://opencode.ai/auth",
|
||||
},
|
||||
},
|
||||
@@ -21,6 +21,9 @@ export default {
|
||||
transport: {
|
||||
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
|
||||
headers: {},
|
||||
usage: {
|
||||
url: "https://opencode.ai/zen/go/v1/usage",
|
||||
},
|
||||
},
|
||||
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
|
||||
// translation. Guarded per-model by `supportedFormats` (see chatCore) because
|
||||
@@ -30,22 +33,41 @@ export default {
|
||||
{ format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } },
|
||||
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
|
||||
],
|
||||
// supportedFormats follow the endpoint table in https://opencode.ai/docs/go/
|
||||
models: [
|
||||
{ id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
|
||||
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
|
||||
{ id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] },
|
||||
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
|
||||
{ id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
|
||||
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
|
||||
{ id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] },
|
||||
{ id: "hy3", name: "Hy3", supportedFormats: ["openai"] },
|
||||
// Served by /zen/go/v1/responses only — the responses-only entry forces chatCore
|
||||
// past the sourceFormat-matched transports into translation (see chatCore guard).
|
||||
{ id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
|
||||
],
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
};
|
||||
|
||||
@@ -17,13 +17,17 @@ export default {
|
||||
headers: {
|
||||
"x-opencode-client": "desktop",
|
||||
},
|
||||
forceStream: true,
|
||||
noAuth: true,
|
||||
quirks: {
|
||||
forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"],
|
||||
},
|
||||
},
|
||||
models: [
|
||||
// Muse Spark models are served by /zen/v1/responses; the rest stay on
|
||||
// /chat/completions, so the format is declared per-model, not per-provider.
|
||||
// Endpoint formats differ per model, so declare non-chat models explicitly.
|
||||
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
|
||||
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
|
||||
{ id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" },
|
||||
],
|
||||
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
|
||||
passthroughModels: true,
|
||||
|
||||
@@ -40,8 +40,11 @@ export default {
|
||||
{ id: "openai/gpt-image-1", name: "GPT Image 1 (via OpenRouter)", params: ["n","size","quality","response_format"], kind: "image" },
|
||||
{ id: "google/imagen-3.0-generate-002", name: "Imagen 3 (via OpenRouter)", params: ["n","size"], kind: "image" },
|
||||
{ id: "black-forest-labs/FLUX.1-schnell", name: "FLUX.1 Schnell (via OpenRouter)", params: ["n","size"], kind: "image" },
|
||||
{ id: "google/veo-3.1", name: "Veo 3.1 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
|
||||
{ id: "openai/sora-2-pro", name: "Sora 2 Pro (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
|
||||
{ id: "bytedance/seedance-2.0", name: "Seedance 2.0 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
|
||||
],
|
||||
serviceKinds: ["llm","embedding","tts","imageToText"],
|
||||
serviceKinds: ["llm","embedding","tts","imageToText","video"],
|
||||
ttsConfig: {
|
||||
baseUrl: "https://openrouter.ai/api/v1/chat/completions",
|
||||
defaultModel: "openai/gpt-4o-mini-tts",
|
||||
@@ -57,6 +60,12 @@ export default {
|
||||
baseUrl: "https://openrouter.ai/api/v1/images/generations",
|
||||
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
|
||||
},
|
||||
// Async video jobs (POST /videos → { id, status }, GET /videos/{id} polls).
|
||||
// Docs: https://openrouter.ai/docs/api/api-reference/videos
|
||||
videoConfig: {
|
||||
baseUrl: "https://openrouter.ai/api/v1/videos",
|
||||
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
|
||||
},
|
||||
modelsFetcher: { url: "https://openrouter.ai/api/v1/models", type: "openrouter-free" },
|
||||
passthroughModels: true,
|
||||
};
|
||||
|
||||
@@ -30,12 +30,15 @@ export default {
|
||||
{ id: "auto", name: "Auto" },
|
||||
{ id: "performance", name: "Performance" },
|
||||
{ id: "efficient", name: "Efficient" },
|
||||
{ id: "qmodel_preview", name: "Qwen3.8-Max-Preview" },
|
||||
{ id: "lite", name: "Lite" },
|
||||
{ id: "qmodel_38max", name: "Qwen3.8-Max" },
|
||||
{ id: "qmodel_latest", name: "Qwen3.7-Max" },
|
||||
{ id: "qmodel", name: "Qwen3.7-Plus" },
|
||||
{ id: "qfmodel", name: "Qwen3.8-Flash" },
|
||||
{ id: "kmodel_latest", name: "Kimi-K3" },
|
||||
{ id: "kmodel", name: "Kimi-K2.7-Code" },
|
||||
{ id: "gm51model", name: "GLM-5.2" },
|
||||
{ id: "gmodel", name: "GLM-5.3" },
|
||||
{ id: "gfmodel", name: "GLM-5.3-Flash" },
|
||||
{ id: "dmodel", name: "DeepSeek-V4-Pro" },
|
||||
{ id: "dfmodel", name: "DeepSeek-V4-Flash" },
|
||||
{ id: "mmodel", name: "MiniMax-M3" },
|
||||
|
||||
@@ -27,6 +27,13 @@ export default {
|
||||
{ id: "gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview" },
|
||||
{ id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
|
||||
{ id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" },
|
||||
{ id: "veo-3.1-generate-preview", name: "Veo 3.1 (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
|
||||
{ id: "veo-3.1-fast-generate-preview", name: "Veo 3.1 Fast (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
|
||||
{ id: "veo-3.0-generate-001", name: "Veo 3", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" },
|
||||
{ id: "veo-2.0-generate-001", name: "Veo 2", params: ["duration","aspect_ratio","negative_prompt","seed","storage_uri"], kind: "video" },
|
||||
],
|
||||
serviceKinds: ["llm","imageToText"],
|
||||
serviceKinds: ["llm","imageToText","video"],
|
||||
// Veo via predictLongRunning + fetchPredictOperation (adapter: handlers/videoProviders/vertex.js).
|
||||
// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation
|
||||
videoConfig: { baseUrl: "https://aiplatform.googleapis.com" },
|
||||
};
|
||||
|
||||
@@ -1,11 +1,19 @@
|
||||
import { CLAUDE_API_HEADERS } from "../shared.js";
|
||||
|
||||
// Dual auth (same pattern as kimi):
|
||||
// - API key (sk-...) → cloud API on api.xiaomimimo.com
|
||||
// - Desktop account/OAuth → same cloud host, plus the Desktop-exclusive Preview
|
||||
// models served by the account-service route on mimo-server-cn.xiaomimimo.com
|
||||
// (authorized by a Xiaomi account session cookie, not the key).
|
||||
// Endpoint is picked per model in the executor, same as opencode-go's /responses split.
|
||||
export default {
|
||||
id: "xiaomi-mimo",
|
||||
priority: 290,
|
||||
alias: "xiaomi-mimo",
|
||||
aliases: [
|
||||
"mimo",
|
||||
"mimo-desktop",
|
||||
"xmd",
|
||||
],
|
||||
uiAlias: "mimo",
|
||||
display: {
|
||||
@@ -16,9 +24,12 @@ export default {
|
||||
website: "https://xiaomimimo.com",
|
||||
notice: {
|
||||
apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys",
|
||||
signupUrl: "https://mimo.xiaomimimo.com/desktop/invite/",
|
||||
},
|
||||
},
|
||||
category: "apikey",
|
||||
category: "oauth",
|
||||
authModes: ["oauth", "apikey"],
|
||||
hasOAuth: true,
|
||||
serviceKinds: ["llm", "tts"],
|
||||
transport: {
|
||||
baseUrl: "https://api.xiaomimimo.com/v1/chat/completions",
|
||||
@@ -39,6 +50,11 @@ export default {
|
||||
},
|
||||
],
|
||||
models: [
|
||||
// Desktop-exclusive — served by the account-service route, which only accepts
|
||||
// OpenAI format, so supportedFormats pins them to the openai transport.
|
||||
{ id: "mimo-x-pro-preview", name: "MiMo-X-Pro-Preview", upstreamModelId: "xiaomi/mimo-x-pro-preview", supportedFormats: ["openai"] },
|
||||
{ id: "mimo-x-flash-preview", name: "MiMo-X-Flash-Preview", upstreamModelId: "xiaomi/mimo-x-flash-preview", supportedFormats: ["openai"] },
|
||||
// Cloud API models (api.xiaomimimo.com/v1)
|
||||
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" },
|
||||
{ id: "mimo-v2.5", name: "MiMo V2.5" },
|
||||
{ id: "mimo-v2-omni", name: "MiMo V2 Omni" },
|
||||
@@ -51,4 +67,18 @@ export default {
|
||||
authHeader: "bearer",
|
||||
format: "xiaomi-mimo-tts",
|
||||
},
|
||||
features: {
|
||||
usage: true,
|
||||
usageApikey: true,
|
||||
},
|
||||
// Custom OAuth — non-standard ECDH encrypted-callback flow.
|
||||
// Handled by the Xiaomi MiMo OAuth service, not the generic PKCE pipeline.
|
||||
oauth: {
|
||||
custom: true,
|
||||
authorizeUrl: "https://platform.xiaomimimo.com/authorize",
|
||||
// The callback carries ?u=<ECDH-encrypted payload> instead of ?code=.
|
||||
// Decryption yields { uid, sk, url }.
|
||||
callbackParam: "u",
|
||||
kn: "mimocode",
|
||||
},
|
||||
};
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
// Zed provider — RSA keypair callback auth (NOT standard OAuth).
|
||||
export default {
|
||||
id: "zed",
|
||||
priority: 10,
|
||||
priority: 999,
|
||||
alias: "zd",
|
||||
uiAlias: "zd",
|
||||
hidden: true,
|
||||
display: {
|
||||
name: "Zed",
|
||||
icon: "code",
|
||||
|
||||
@@ -62,8 +62,17 @@ const ANTHROPIC_BETA_BASE = [
|
||||
const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"];
|
||||
|
||||
// Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them.
|
||||
export function selectAnthropicBeta(model = "") {
|
||||
const flags = [...ANTHROPIC_BETA_BASE];
|
||||
// `redact-thinking` asks Anthropic to return signature-only thinking blocks, which
|
||||
// is right for clients that never render thinking but blanks the summaries a
|
||||
// client explicitly requested with `thinking.display: "summarized"`.
|
||||
const ANTHROPIC_BETA_REDACT_THINKING = "redact-thinking-2026-02-12";
|
||||
|
||||
export function wantsThinkingSummaries(body) {
|
||||
return body?.thinking?.display === "summarized";
|
||||
}
|
||||
|
||||
export function selectAnthropicBeta(model = "", body = null) {
|
||||
const flags = ANTHROPIC_BETA_BASE.filter((flag) => flag !== ANTHROPIC_BETA_REDACT_THINKING || !wantsThinkingSummaries(body));
|
||||
if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT);
|
||||
return flags.join(",");
|
||||
}
|
||||
|
||||
@@ -26,6 +26,7 @@ const FORMAT_LEVELS = {
|
||||
qwen: L.base,
|
||||
kimi: L.levelMax,
|
||||
deepseek: L.hiMax,
|
||||
commandcode: ["none", "low", "medium", "high", "xhigh", "max"],
|
||||
minimax: L.onOff,
|
||||
hunyuan: L.base,
|
||||
step: L.base,
|
||||
@@ -35,17 +36,29 @@ const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh
|
||||
|
||||
// Model-name pattern overrides (glob, first match wins) — more precise than format default.
|
||||
const PATTERN_THINKING = [
|
||||
{ provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] },
|
||||
{ provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS },
|
||||
{ pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking
|
||||
// codebuddy-cn per-model effort sets — read off the client picker (server-
|
||||
// delivered supportedEfforts), 2026-08-30. Gateway uses thinkingFormat "openai"
|
||||
// but rejects levels outside each model's set.
|
||||
// DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max
|
||||
// all 200 via output_config.effort; "none" is a 400 on the anthropic route
|
||||
// (disable thinking instead). none kept for the picker = disable.
|
||||
{ pattern: "*deepseek-v4.*", levels: ["none", "low", "medium", "high", "xhigh", "max"] },
|
||||
// codebuddy-cn per-model effort sets — the server's product-config payload
|
||||
// publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint
|
||||
// accepts any level you send (probed none/minimal/low/medium/high/xhigh/max
|
||||
// → all 200), but values outside a model's supportedEfforts are silently
|
||||
// clamped, so the declared set stays authoritative for the picker. Models
|
||||
// that publish no supportedEfforts (glm-5.1 / glm-5v-turbo / kimi-k2.x /
|
||||
// kimi-k3-1 / minimax-m3) fall through to the openai format default.
|
||||
{ provider: "codebuddy-cn", pattern: "glm-5.3*", levels: ["low", "high", "max"] },
|
||||
{ provider: "codebuddy-cn", pattern: "glm-5.2", levels: ["high", "xhigh"] },
|
||||
{ provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
|
||||
{ provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] },
|
||||
{ provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] },
|
||||
// codebuddy-intl rides the same gateway catalog, so its deepseek levels match.
|
||||
{ provider: "codebuddy-intl", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] },
|
||||
];
|
||||
|
||||
// Returns valid thinking levels for a model, or null when the model has no reasoning.
|
||||
|
||||
@@ -13,7 +13,7 @@ export function injectSystemPrompt(body, format, prompt) {
|
||||
if (!body || !prompt) return;
|
||||
if (typeof body !== "object") return;
|
||||
|
||||
// Kiro wire shape is unique (conversationState/systemPrompt) — handle directly.
|
||||
// Kiro wire shape is unique (conversationState) — handle directly.
|
||||
if (isKiroBody(body) || format === FORMATS.KIRO) {
|
||||
injectKiroSystem(body, prompt);
|
||||
return;
|
||||
@@ -61,10 +61,13 @@ export function injectSystemPrompt(body, format, prompt) {
|
||||
|
||||
function isKiroBody(body) {
|
||||
if (!body || typeof body !== "object") return false;
|
||||
if (typeof body.systemPrompt !== "string") return false;
|
||||
const cs = body.conversationState;
|
||||
if (!cs || typeof cs !== "object") return false;
|
||||
return Array.isArray(cs.history) || !!(cs.currentMessage && typeof cs.currentMessage === "object");
|
||||
// A top-level `systemPrompt` used to be the marker, but the Kiro translator no
|
||||
// longer emits it (kiro.dev rejects the field), so gate on the turn shape.
|
||||
const historyTurn = Array.isArray(cs.history)
|
||||
&& cs.history.some(it => it && (it.userInputMessage || it.assistantResponseMessage));
|
||||
return historyTurn || !!(cs.currentMessage && cs.currentMessage.userInputMessage);
|
||||
}
|
||||
|
||||
// Exact idempotency: prompt present as its own SEP-delimited segment (or the
|
||||
@@ -258,80 +261,33 @@ function injectGeminiSystem(body, prompt) {
|
||||
}
|
||||
|
||||
// ---- Kiro ----
|
||||
// Updates top-level systemPrompt and only the mirrored leading prefix of the
|
||||
// first user history turn, else current user. next = old + SEP + prompt.
|
||||
// Replace old leading prefix only; preserve time context and user tail.
|
||||
// The prompt is appended to the first user turn's content — the same place the
|
||||
// Kiro translator already mirrors the system text via its contentPrefix.
|
||||
//
|
||||
// A top-level `systemPrompt` is deliberately NOT written: the kiro.dev gateway
|
||||
// answers any body carrying that field with
|
||||
// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"}
|
||||
// The translator stopped emitting it in v0.5.59, but this injector kept adding
|
||||
// it back, so every kr/ model failed whenever an RTK prompt (caveman, ponytail)
|
||||
// was active.
|
||||
function injectKiroSystem(body, prompt) {
|
||||
try {
|
||||
let oldPrompt = typeof body.systemPrompt === "string" ? body.systemPrompt : "";
|
||||
// Repair path: a previous partial write left systemPrompt updated but user
|
||||
// content still mirroring the pre-write prefix. Re-derive the effective old
|
||||
// prefix from content so this pass converges instead of early-returning.
|
||||
const cs0 = body.conversationState;
|
||||
let firstUser0 = cs0 && Array.isArray(cs0.history)
|
||||
? (cs0.history.find(it => it && it.userInputMessage)?.userInputMessage ?? null)
|
||||
: null;
|
||||
if (!firstUser0 && cs0?.currentMessage?.userInputMessage) firstUser0 = cs0.currentMessage.userInputMessage;
|
||||
|
||||
if (firstUser0 && typeof firstUser0.content === "string" && oldPrompt && !hasPrompt(oldPrompt, prompt)) {
|
||||
const c0 = firstUser0.content;
|
||||
if (c0 === oldPrompt || (c0.startsWith(oldPrompt) && !c0.startsWith(`${oldPrompt}${SEP}`))) {
|
||||
// systemPrompt advanced past mirrored prefix → stale; treat as un-mirrored
|
||||
oldPrompt = "";
|
||||
}
|
||||
}
|
||||
if (oldPrompt && hasPrompt(oldPrompt, prompt)) return;
|
||||
const next = oldPrompt ? `${oldPrompt}${SEP}${prompt}` : prompt;
|
||||
|
||||
// Atomicity: write user content first, then systemPrompt only if content
|
||||
// write succeeded (or was a no-op). If systemPrompt write then fails, the
|
||||
// repair heuristic above re-derives from content on retry — no permanent
|
||||
// half-applied state.
|
||||
const cs = body.conversationState;
|
||||
let targetMsg = null;
|
||||
try {
|
||||
const hist = Array.isArray(cs?.history) ? cs.history : null;
|
||||
if (hist) {
|
||||
for (const item of hist) {
|
||||
if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; }
|
||||
}
|
||||
}
|
||||
if (!targetMsg && cs?.currentMessage?.userInputMessage) {
|
||||
targetMsg = cs.currentMessage.userInputMessage;
|
||||
}
|
||||
} catch (_) { targetMsg = null; }
|
||||
|
||||
let sysWritten = false;
|
||||
try { body.systemPrompt = next; sysWritten = true; } catch (_) {}
|
||||
|
||||
const applyContent = () => {
|
||||
const content = typeof targetMsg.content === "string" ? targetMsg.content : "";
|
||||
if (oldPrompt === "") {
|
||||
// Empty old prompt: prepend unless already at head (exact, not substring)
|
||||
if (content.startsWith(prompt) || content.startsWith(next)) return;
|
||||
const newContent = content ? `${next}${SEP}${content}` : next;
|
||||
try { targetMsg.content = newContent; } catch (_) {}
|
||||
return;
|
||||
}
|
||||
if (!content.startsWith(oldPrompt)) return; // not mirrored at head — leave alone
|
||||
if (content.startsWith(next)) return; // already applied → idempotent
|
||||
const tail = content.slice(oldPrompt.length);
|
||||
try { targetMsg.content = `${next}${tail}`; } catch (_) {}
|
||||
};
|
||||
|
||||
try {
|
||||
if (targetMsg) applyContent();
|
||||
} catch (_) {}
|
||||
if (sysWritten && targetMsg) {
|
||||
// verify convergence: content should now start with next (or be un-mirrored)
|
||||
let ok = false;
|
||||
try {
|
||||
const c = targetMsg.content;
|
||||
ok = typeof c !== "string" || c.startsWith(next) || !c.startsWith(oldPrompt);
|
||||
} catch (_) {}
|
||||
if (!ok) {
|
||||
try { body.systemPrompt = oldPrompt; } catch (_) {} // rollback
|
||||
const hist = Array.isArray(cs?.history) ? cs.history : null;
|
||||
if (hist) {
|
||||
for (const item of hist) {
|
||||
if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; }
|
||||
}
|
||||
}
|
||||
if (!targetMsg && cs?.currentMessage?.userInputMessage) {
|
||||
targetMsg = cs.currentMessage.userInputMessage;
|
||||
}
|
||||
if (!targetMsg) return;
|
||||
|
||||
const content = typeof targetMsg.content === "string" ? targetMsg.content : "";
|
||||
const next = dedupStringAppend(content, prompt);
|
||||
if (next === content) return; // already injected — idempotent across retries
|
||||
try { targetMsg.content = next; } catch (_) { /* frozen/proxy fail-open */ }
|
||||
} catch (_) {}
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user