merge: sync upstream v0.5.81 into MIBP fork

# Conflicts:
#	.gitignore
#	Dockerfile
#	open-sse/handlers/chatCore.js
#	open-sse/providers/registry/cline.js
#	open-sse/providers/registry/index.js
#	open-sse/services/usage.js
#	open-sse/utils/streamHandler.js
#	package.json
#	src/app/(dashboard)/dashboard/profile/page.js
#	src/app/(dashboard)/dashboard/providers/[id]/page.js
This commit is contained in:
MUH. IQRAM BAHRING
2026-09-19 11:35:34 +08:00
208 changed files with 13062 additions and 977 deletions
+12
View File
@@ -89,3 +89,15 @@ graphify-out/*
# Kiro local workspace state
.kiro/
9router-*
# Local sensitive / temp files
.engine-token.txt
.tmp-prov.json
.tmp-providers.json
.oauth-session.json
.dev-server.log
.dev-server-err.log
.start-dev.ps1
start-dev-silent.cjs
debug.log
+68
View File
@@ -1,3 +1,71 @@
# v0.5.81 (2026-09-18)
## Features
- **Xiaomi MiMo**: merge MiMo Desktop support into `xiaomi-mimo` with dual auth (API key + Desktop/OAuth session), Preview models support, and encrypted-callback OAuth flow
- **Claude Code**: add 1M-context toggle (`[1m]` marker) and drive `CLAUDE_CODE_AUTO_COMPACT_WINDOW` directly from the dashboard
- **Models**: add DeepSeek-V4.1-Flash to DeepSeek provider, CodeBuddy-Intl, and Ollama (`deepseek-v4.1-flash:cloud`); enable `low`..`max` reasoning effort levels and vision capability for DeepSeek-V4.*
- **i18n**: integrate Persian (fa) translation
## Fixes
- **OpenCode / OpenCode Go**: resolve 403 `FreeTierError` and 429 rate limits with canonical session format, valid User-Agent, and stable upstream session reuse; force stream and declare `forceStream` for free-tier SSE aggregation; cloak decoy tools, normalize Muse Free tool choice, and strip prior reasoning items on Responses models; route Union Alpha via Messages API
- **Kiro**: preserve underscores in tool names (`mcp__server__tool`) and restore client tool names in responses; use neutral placeholder for tool-result-only turns; forward tool-result images
- **Stream**: report aborts after HTTP 200 in-band (per-format error frames) instead of closing silently
- **Command Code**: preserve images and `reasoning_effort` on `/alpha/generate`; retry transient stream errors and avoid fake stop chunks; add Quota Tracker support
- **Zed**: harden OAuth lifecycle (preserve `systemId`, renew proxy timeout), support live model resolution, and lower display priority in OAuth list
- **Antigravity**: scope cached thought signatures to model family; strip Claude Code billing headers from system prompts; sanitize Hermes system identity
- **Codex**: route bare `codex-auto-review` requests to the Codex provider (#4135)
- **Auth**: do not cool down an account for request-scoped 4xx errors
- **Usage**: improve DeepSeek credit balance display as currency credit instead of 0/total quota bar
- **Model Catalog**: scope synced catalog to gateways and declare vision capabilities for DeepSeek V4.1-Flash IDs
# v0.5.75 (2026-09-10)
## Features
- **Video**: add OpenRouter and Vertex AI (Veo) video generation on `/v1/videos/*` via a provider adapter layer; poll requests resolve their provider from `x-connection-id` or `?provider=`
- **Antigravity**: add weekly quota tracking (Gemini weekly / Claude & GPT weekly) and free-tier handling from `retrieveUserQuotaSummary` (#3892)
- **Codex**: add GPT Image 2.5, Flare and Sunburst image models with multi-image support; add the same ids to the OpenAI catalog
- **Qoder**: surface usage to all clients and stop inlining large attachments — images upload through `/api/v2/image/upload` like qodercli, oversized file blocks become stubs, context tier auto-escalates
- **OpenCode Go**: add newly published models (glm-5.3, kimi-k3, deepseek-flash, longcat-2.0, hy4-preview, hy3 on chat/completions; qwen3.8-max, qwen3.8-flash on `/messages`; grok-4.6, gpt-5.6-luna on Responses) and list `deepseek-v4.1-flash` first in the catalog
- **CLI tools**: group the model selector by provider with full-text search and manual custom model ID entry
- **CodeBuddy-CN**: replace `deepseek-v4-flash` with `deepseek-v4.1-flash`
## Fixes
- **Tools**: scope Claude tool type defaulting to gateways declaring `requireClaudeToolType` — the global default broke Anthropic-compatible endpoints that only accept the legacy typeless tool shape (#3905)
- **Claude**: cap re-anchored `cache_control` at the 4-marker budget so a spent budget no longer 400s and triggers a full combo failover; wrap bare single-object content turns before the mid-conversation-system fold
- **Cline / Airforce**: unwrap the `{"success":true,"data":…}` envelope on non-stream chat completions (#3644); add the live Cline/ClinePass model catalog and refresh Airforce free models
- **Cline**: stop `workos:`-prefixing ClinePass API keys (401 on every request, #2333) and add clinepass token refresh
- **Kiro**: never send a top-level `systemPrompt` (`400 REQUEST_BODY_INVALID`); route requests through current runtime surfaces (#3776)
- **Codex**: strip Unicode-property tool schema patterns the validator rejects (#3922); restore the `Version` header and single-source the CLI version
- **DeepSeek**: keep Anthropic-only tool types when forwarding to `/anthropic/v1/messages`
- **Qoder**: drop the Responses usage plumbing from shared translator/handler code, which changed token accounting for every provider, not just Qoder
- **Antigravity**: normalize contents and handle intermediate tool responses; protect the OAuth token-refresh path from Google anti-abuse rate limits (#3813)
- **Providers**: clear stale connection health state (`modelLock_*`, `backoffLevel`, `rateLimitedUntil`, `errorCode`) when a connection is re-validated (#3810, #3830); remove the duplicate `qwen` provider that shadowed `alims-intl`
- **Video / Vertex**: reject job ids and model ids that would escape the request URL path (SSRF)
- **Usage**: parse the Fable weekly limit from `limits[]` instead of fabricating a row (#3847)
- **Auth**: set a 24h `maxAge` on the dashboard session cookie
# v0.5.69 (2026-09-05)
## Features
- **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities
- **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`)
- **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys
- **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819)
- **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through
- **CLI tools**: replace Copilot MITM with VS Code extension setup guide
- **Gemini**: persist and replay `thoughtSignature` scoped by session namespace
## Fixes
- **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792)
- **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813)
- **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797)
- **Dashboard**: dynamic mode label for local/remote detection (#3801)
- **Codex**: format reset credit API errors cleanly (#3778)
- **Security**: guard cowork MCP tools probe against SSRF (#3783)
- **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800)
- **Logger**: suppress noisy background token refresh logs
- **CLI**: export packed `.tgz` directly into workspace root instead of parent directory
# v0.5.65 (2026-09-03)
## Features
+2 -2
View File
@@ -1,6 +1,6 @@
{
"name": "9router",
"version": "0.5.65",
"version": "0.5.81",
"description": "9Router CLI - Start and manage 9Router server",
"bin": {
"9router": "./cli.js"
@@ -16,7 +16,7 @@
"scripts": {
"dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js",
"build": "node scripts/build-cli.js",
"pack:cli": "npm run build && npm pack --pack-destination ../..",
"pack:cli": "npm run build && npm pack --pack-destination ..",
"publish:cli": "npm run build && npm publish",
"postinstall": "node hooks/postinstall.js",
"prepublishOnly": "npm run build"
+182 -45
View File
@@ -55,7 +55,7 @@ async function getAvailableModelsGrouped() {
}
/**
* Display model list and prompt for selection
* Display model list and prompt for selection with provider grouping & search
* @param {string} title - Title to display
* @param {string} currentValue - Current selected value (optional)
* @param {Object} options - { excludeCombos?: boolean }
@@ -70,62 +70,199 @@ async function selectModelFromList(title, currentValue = "", options = {}) {
if (totalModels === 0) {
return null;
}
// Build flat list for selection
const allModels = [];
// Display
clearScreen();
console.log(`\n🎯 ${title}`);
console.log("=".repeat(50));
if (currentValue) {
console.log(`Current: ${currentValue}\n`);
} else {
console.log();
}
let idx = 1;
// Combos first (skipped when excludeCombos is true)
// All models for flat search
const allModelsList = [
...combos,
...Object.values(groups).flat()
];
// Build category list
const categories = [];
if (combos.length > 0) {
console.log("[Combos]");
combos.forEach(combo => {
console.log(` ${idx}. ${combo}`);
allModels.push(combo);
idx++;
categories.push({
id: "combos",
name: "[Combos]",
models: combos
});
console.log();
}
// Provider groups in order (by alias)
const sortedProviders = Object.keys(groups).sort((a, b) => {
const idxA = PROVIDER_ALIAS_ORDER.indexOf(a);
const idxB = PROVIDER_ALIAS_ORDER.indexOf(b);
return (idxA === -1 ? 999 : idxA) - (idxB === -1 ? 999 : idxB);
});
sortedProviders.forEach(provider => {
sortedProviders.forEach((provider) => {
const providerName = PROVIDER_ALIAS_NAMES[provider] || provider;
console.log(`[${providerName}]`);
groups[provider].forEach(model => {
console.log(` ${idx}. ${model}`);
allModels.push(model);
idx++;
categories.push({
id: provider,
name: providerName,
models: groups[provider]
});
console.log();
});
console.log(" 0. Cancel\n");
// Prompt for number input
const input = await prompt("Enter number: ");
const num = parseInt(input, 10);
if (isNaN(num) || num === 0 || num < 0 || num > allModels.length) {
return null;
let filterQuery = null;
while (true) {
clearScreen();
console.log(`\n🎯 ${title}`);
console.log("=".repeat(50));
if (currentValue) {
console.log(`Current: ${currentValue}\n`);
} else {
console.log();
}
// Active search view
if (filterQuery !== null) {
const q = filterQuery.toLowerCase().trim();
const matched = allModelsList.filter((m) => m.toLowerCase().includes(q));
console.log(`🔍 Search results for "${filterQuery}": (${matched.length} found)\n`);
if (matched.length === 0) {
console.log(" No matching models found.\n");
console.log(" 0. ← Back to providers");
console.log(" s. Search again\n");
const act = await prompt("Select option: ");
if (act.toLowerCase() === "s") {
const newQ = await prompt("Enter search keyword: ");
filterQuery = newQ.trim() || null;
} else {
filterQuery = null;
}
continue;
}
matched.forEach((m, i) => {
console.log(` ${i + 1}. ${m}`);
});
console.log("\n 0. ← Back to providers");
console.log(" s. Search again\n");
const input = await prompt("Enter number to select (or 0/s): ");
if (input.toLowerCase() === "s") {
const newQ = await prompt("Enter search keyword: ");
filterQuery = newQ.trim() || null;
continue;
}
const num = parseInt(input, 10);
if (isNaN(num) || num === 0) {
filterQuery = null;
continue;
}
if (num > 0 && num <= matched.length) {
return matched[num - 1];
}
continue;
}
// If only 1 category exists, jump straight into its model list
if (categories.length === 1) {
const singleCategory = categories[0];
console.log(`[${singleCategory.name}]`);
singleCategory.models.forEach((m, i) => {
console.log(` ${i + 1}. ${m}`);
});
console.log();
console.log(" s. 🔍 Search models");
console.log(" m. ✍️ Enter custom model ID");
console.log(" 0. Cancel\n");
const input = await prompt("Enter choice (number / s / m / 0): ");
const trimmed = input.trim();
if (!trimmed || trimmed === "0") return null;
const lower = trimmed.toLowerCase();
if (lower === "s") {
const q = await prompt("Enter search keyword: ");
if (q.trim()) filterQuery = q.trim();
continue;
}
if (lower === "m") {
const customModel = await prompt("Enter custom model ID: ");
if (customModel.trim()) return customModel.trim();
continue;
}
const num = parseInt(trimmed, 10);
if (!isNaN(num) && num > 0 && num <= singleCategory.models.length) {
return singleCategory.models[num - 1];
}
filterQuery = trimmed;
continue;
}
// Multiple categories view
console.log("[Providers & Groups]");
categories.forEach((cat, i) => {
console.log(` ${i + 1}. ${cat.name} (${cat.models.length} models)`);
});
console.log();
console.log(" s. 🔍 Search models");
console.log(" m. ✍️ Enter custom model ID");
console.log(" 0. Cancel\n");
const input = await prompt("Enter choice (number / keyword / s / m): ");
const trimmed = input.trim();
if (!trimmed || trimmed === "0") {
return null;
}
const lower = trimmed.toLowerCase();
if (lower === "s") {
const q = await prompt("Enter search keyword: ");
if (q.trim()) {
filterQuery = q.trim();
}
continue;
}
if (lower === "m") {
const customModel = await prompt("Enter custom model ID: ");
if (customModel.trim()) {
return customModel.trim();
}
continue;
}
const num = parseInt(trimmed, 10);
// Selected a category
if (!isNaN(num) && num > 0 && num <= categories.length) {
const selectedCategory = categories[num - 1];
while (true) {
clearScreen();
console.log(`\n🎯 ${title} > ${selectedCategory.name}`);
console.log("=".repeat(50));
if (currentValue) {
console.log(`Current: ${currentValue}\n`);
} else {
console.log();
}
selectedCategory.models.forEach((m, i) => {
console.log(` ${i + 1}. ${m}`);
});
console.log("\n 0. ← Back\n");
const modelChoice = await prompt("Enter number to select (0 to back): ");
const modelNum = parseInt(modelChoice, 10);
if (isNaN(modelNum) || modelNum === 0) {
break;
}
if (modelNum > 0 && modelNum <= selectedCategory.models.length) {
return selectedCategory.models[modelNum - 1];
}
}
continue;
}
// User typed text directly -> treat as search query
filterQuery = trimmed;
}
return allModels[num - 1];
}
module.exports = {
@@ -0,0 +1,261 @@
# OpenCode Go Session Header Implementation Plan
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
**Goal:** Send a stable, conversation-scoped `x-opencode-session` header on every OpenCode Go request and install the patched CLI locally.
**Architecture:** Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. `chatCore` passes the provider-scoped session resolved from the original request plus the detected client tool; the executor derives a request-local upstream session and delegates all existing transport, authentication, retry, and proxy behavior to `DefaultExecutor`.
**Tech Stack:** Node.js ESM, Vitest, Next.js, npm CLI packaging, GitHub CLI.
## Global Constraints
- Apply the header to OpenCode Go chat completions, Claude Messages, and OpenAI Responses transports.
- Preserve a valid native `x-opencode-session`; hash all translated non-OpenCode identities to `ses_<32 lowercase hex>`.
- Namespace translated identities by detected client tool, using `generic` when unknown.
- Do not keep mutable per-request session state on the executor singleton or mutate the caller's credentials object.
- Do not change OpenCode Go models, routing, reasoning, tool behavior, dependencies, or unrelated providers.
- Reuse upstream issue #3759 instead of creating a duplicate issue.
---
### Task 1: Add Failing OpenCode Go Session Tests
**Files:**
- Create: `tests/unit/opencode-go-session.test.js`
**Interfaces:**
- Consumes: `getExecutor(provider)` and `DefaultExecutor.buildHeaders(credentials, stream, url, model)`.
- Produces: the required public behavior for `OpenCodeGoExecutor.prepareRequestCredentials({ body, credentials, providerSessionId, clientTool })` and `OpenCodeGoExecutor.execute(args)`.
- [ ] **Step 1: Write the failing tests**
Create a Vitest suite that mocks `proxyAwareFetch`, obtains `getExecutor("opencode-go")`, and asserts:
```js
const prepared = executor.prepareRequestCredentials({
body: { messages: [{ role: "user", content: "hello" }] },
credentials: { apiKey: "test-key", connectionId: "conn-a", rawHeaders: {} },
providerSessionId: "conversation-a",
clientTool: "claude",
});
expect(prepared).not.toBe(credentials);
expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/);
expect(credentials).not.toHaveProperty("_opencodeGoSession");
```
Cover native header preservation, stable values across all three runtime transports, different conversation IDs, different client tools using the same ID, connection fallback, no singleton state, no header on `DefaultExecutor("openai")`, and the final fetch headers returned by `execute()`.
- [ ] **Step 2: Run the focused test and verify RED**
Run:
```bash
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
```
Expected: FAIL because `getExecutor("opencode-go")` still returns `DefaultExecutor` and `prepareRequestCredentials` does not exist.
- [ ] **Step 3: Commit the failing test**
```bash
git add tests/unit/opencode-go-session.test.js
git commit -m "test: cover OpenCode Go session headers"
```
### Task 2: Implement the Dedicated Executor
**Files:**
- Create: `open-sse/executors/opencode-go.js`
- Modify: `open-sse/executors/index.js`
**Interfaces:**
- Consumes: `DefaultExecutor`, `resolveSessionId()`, request `credentials.rawHeaders`, `providerSessionId`, and `clientTool`.
- Produces: `OpenCodeGoExecutor`, `prepareRequestCredentials()`, and an `execute()` override that delegates with cloned credentials.
- [ ] **Step 1: Add the minimal executor implementation**
Implement these rules:
```js
function translatedSessionId(sessionId, clientTool) {
const digest = crypto
.createHash("sha256")
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
.digest("hex")
.slice(0, 32);
return `ses_${digest}`;
}
```
`prepareRequestCredentials()` must read a case-insensitive native
`x-opencode-session` with the same non-empty, 256-character cap used by the
session manager. Otherwise it uses `providerSessionId` or calls
`resolveSessionId({ headers, body, connectionId, scope: "opencode-go" })`, then
returns `{ ...credentials, _opencodeGoSession: value }`.
`execute(args)` must call `prepareRequestCredentials(args)` and delegate using
`super.execute({ ...args, credentials: prepared })`. `buildHeaders()` must call
`super.buildHeaders()` and add the prepared session, with a connection-scoped
fallback for direct callers.
Register `new OpenCodeGoExecutor()` under `"opencode-go"` and export the class.
- [ ] **Step 2: Run the focused test and verify partial GREEN**
Run:
```bash
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
```
Expected: executor-level tests pass; any chatCore-context assertion remains failing until Task 3.
- [ ] **Step 3: Commit the executor**
```bash
git add open-sse/executors/opencode-go.js open-sse/executors/index.js tests/unit/opencode-go-session.test.js
git commit -m "fix(opencode-go): add stable session header executor"
```
### Task 3: Pass Original Request Session Context
**Files:**
- Modify: `open-sse/handlers/chatCore.js`
- Modify: `tests/unit/opencode-go-session.test.js`
**Interfaces:**
- Consumes: existing `sessionSeed` and `clientTool` variables in `handleChatCore()`.
- Produces: `providerSessionId` and `clientTool` fields on both initial and refreshed-credential calls to `executor.execute()`.
- [ ] **Step 1: Add or enable the failing integration assertion**
Use a mocked executor or source request containing a body-only `session_id` and
assert the executor receives the provider-scoped session resolved before
translation.
- [ ] **Step 2: Run the focused test and verify RED**
Run:
```bash
npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js
```
Expected: FAIL because `handleChatCore()` does not pass `providerSessionId` or
`clientTool` to `executor.execute()`.
- [ ] **Step 3: Pass the request context**
Add the same fields to both executor calls:
```js
executor.execute({
model,
body: translatedBody,
stream,
credentials,
providerSessionId: sessionSeed,
clientTool,
signal: streamController.signal,
log,
proxyOptions,
});
```
- [ ] **Step 4: Run focused and neighboring tests**
Run:
```bash
npx vitest run --config tests/vitest.config.js \
tests/unit/opencode-go-session.test.js \
tests/unit/opencode-go-models.test.js \
tests/unit/session-manager.test.js \
tests/unit/executor-const-guard.test.js
```
Expected: PASS with zero failed tests.
- [ ] **Step 5: Commit the context wiring**
```bash
git add open-sse/handlers/chatCore.js tests/unit/opencode-go-session.test.js
git commit -m "fix(chat): forward provider session context"
```
### Task 4: Verify and Install the Local CLI Package
**Files:**
- Generated: `9router-0.5.65.tgz`
- Packaged output: `cli/app/server.js`
**Interfaces:**
- Consumes: completed source changes and existing CLI build scripts.
- Produces: a globally installed patched `9router@0.5.65`.
- [ ] **Step 1: Run source verification**
```bash
git diff --check origin/master...HEAD
npx vitest run --config tests/vitest.config.js tests/unit/
npm run build
```
Expected: every command exits zero. Record any pre-existing full-suite failures
separately rather than hiding them.
- [ ] **Step 2: Build and package the CLI**
```bash
npm --prefix cli run build
npm --prefix cli pack -- --pack-destination ..
```
Expected: `9router-0.5.65.tgz` exists and contains the patched bundled server.
- [ ] **Step 3: Replace the global npm installation**
```bash
npm install -g ./9router-0.5.65.tgz
```
Expected: `/opt/homebrew/lib/node_modules/9router/package.json` reports `0.5.65`
and the installed bundle contains `x-opencode-session` plus the new executor.
- [ ] **Step 4: Commit any required package-source adjustment**
Do not commit generated tarballs or CLI build artifacts unless the repository
already tracks and requires them.
### Task 5: Publish the Upstream Pull Request
**Files:**
- No additional source files unless verification finds a required correction.
**Interfaces:**
- Consumes: verified branch commits and GitHub issue #3759.
- Produces: a fork branch and a PR against `decolua/9router:master`.
- [ ] **Step 1: Create or repair the GitHub fork remote**
Use `gh repo fork decolua/9router --remote` if the current `fork` remote remains
missing, then push `fix/opencode-go-session-header`.
- [ ] **Step 2: Create the PR**
Use title:
```text
fix(opencode-go): send stable session header
```
The body must include the root cause, downstream-session translation policy,
three covered transports, concurrency behavior, verification evidence,
`Fixes #3759`, and a note that this PR is intentionally narrower than #3780.
- [ ] **Step 3: Verify the published PR**
Run `gh pr view --json number,title,state,url,headRefName,baseRefName` and report
the issue and PR URLs.
@@ -0,0 +1,114 @@
# OpenCode Go Session Header Design
## Problem
OpenCode Go will begin rejecting some requests without an
`x-opencode-session` header on September 6, 2026. In 9Router v0.5.65,
`opencode-go` uses `DefaultExecutor`, whose generic header builder does not add
that header. The specialized OpenCode Free executor already sends it, but that
logic does not apply to the paid OpenCode Go provider or its three transports.
## Goals
- Add `x-opencode-session` to every OpenCode Go chat, Claude Messages, and
OpenAI Responses request.
- Translate a downstream conversation identity into a stable upstream identity.
- Keep identities isolated across different downstream agents and conversations.
- Avoid exposing non-OpenCode downstream session identifiers to OpenCode Go.
- Avoid mutable session state on the shared executor singleton.
- Leave OpenCode Free and all unrelated providers unchanged.
## Non-Goals
- Inferring an exact conversation boundary when a downstream client provides no
session or conversation identifier.
- Adding or changing OpenCode Go models, routing, reasoning, or tool behavior.
- Changing the general session-resolution policy for other providers.
## Architecture
Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. The executor
keeps the existing generic URL, authentication, translation, retry, and proxy
behavior, and overrides only the OpenCode Go session-header concern.
`handleChatCore` already resolves a provider-scoped session from the original
request before translation. It will pass that value and the detected client
tool to `executor.execute()` as request context. `OpenCodeGoExecutor.execute()`
will create a shallow request-local credentials object containing the resolved
OpenCode Go session. It will then delegate to `DefaultExecutor.execute()`.
This avoids storing request state on the executor singleton or mutating shared
provider credentials.
## Session Resolution
The original downstream request remains the source of truth. Existing
`resolveSessionId()` behavior recognizes Claude Code, Antigravity, generic
session headers, and common body fields before request translation can discard
them.
Resolution rules:
1. If the downstream request supplies `x-opencode-session`, treat it as an
authoritative OpenCode identity after trimming and length validation.
2. Otherwise use the provider-scoped session resolved from the original request.
3. Namespace the resolved value with the detected downstream agent, falling back
to `generic` when the agent is unknown.
4. Convert the namespaced value to an opaque deterministic identifier:
`ses_` plus the first 32 hexadecimal characters of SHA-256.
5. If no explicit downstream identity exists, the existing provider connection
fallback guarantees that a header is still sent. It is stable but cannot
distinguish multiple conversations sharing that connection.
The same input conversation produces the same upstream identifier for all three
OpenCode Go transports. Different agents using the same raw session value
produce different identifiers.
## Header Injection
`OpenCodeGoExecutor.buildHeaders()` delegates to
`DefaultExecutor.buildHeaders()` and adds only:
```text
x-opencode-session: <stable-session-id>
```
The implementation applies to:
- `https://opencode.ai/zen/go/v1/chat/completions`
- `https://opencode.ai/zen/go/v1/messages`
- `https://opencode.ai/zen/go/v1/responses`
## Error Handling
Session derivation must not make requests fail. Invalid or oversized native
header values are ignored and the normal resolved-session fallback is used.
Hashing uses Node's built-in `crypto` module and requires no new dependency.
## Testing
Add a focused unit suite that proves:
- all three OpenCode Go transports receive the header;
- the same conversation remains stable across requests and transports;
- different conversations produce different values;
- different agents using the same raw ID remain isolated;
- non-OpenCode session IDs are represented as opaque `ses_<32 hex>` values;
- a valid native `x-opencode-session` remains stable;
- headerless requests still receive a stable fallback;
- OpenCode Free behavior is unchanged;
- unrelated `DefaultExecutor` providers do not receive the header;
- no request state is retained on the shared executor instance.
Run the focused unit tests first, then the neighboring executor/session tests,
the full offline test suite, the application build, and the CLI package build.
## Delivery
Build the CLI with `npm --prefix cli run build`, create a package with
`npm --prefix cli pack`, and install the generated tarball globally to replace
the current npm-installed `9router@0.5.65`. Verify the installed package version
and packaged source contains the new executor.
Upstream issue #3759 already tracks the problem, so no duplicate issue will be
created. The pull request will be narrowly scoped to this fix, reference
`Fixes #3759`, and explain how it differs from the broader open PR #3780.
@@ -111,6 +111,27 @@ Model: cx/gpt-5.2-codex
| `cx/gpt-5.2` | GPT 5.2 | General tasks |
| `cx/gpt-5.1-codex` | GPT 5.1 Codex | Stable coding |
### Image Generation
The Codex image catalog includes `cx/gpt-5.6-sol-image`,
`cx/gpt-5.6-terra-image`, and `cx/gpt-5.6-luna-image`, alongside the existing
GPT 5.5, 5.4, and 5.3 image aliases. Select them under **Image → OpenAI Codex**
in the dashboard, or discover them with `GET /v1/models/image` after connecting
a Codex account.
```bash
curl http://localhost:20128/v1/images/generations \
-H "Authorization: Bearer $NINE_ROUTER_API_KEY" \
-H "Content-Type: application/json" \
-d '{"model":"cx/gpt-5.6-sol-image","prompt":"A blue square","size":"1024x1024"}'
```
These are 9Router aliases: the image adapter removes `-image` and sends the
underlying model an `image_generation` tool through the Codex Responses API.
The same endpoint accepts an `image` reference for edits. Image generation
requires an eligible ChatGPT Plus or higher account; availability of each
underlying model and its image tool depends on the connected account.
### Pro Tips
- **5-hour rolling quota** - Fresh quota every 5 hours
+8
View File
@@ -7,6 +7,9 @@ import { createRequire } from "module";
export const GEMINI_CLI_VERSION = PROVIDERS["gemini-cli"]?.cliVersion;
export const GEMINI_CLI_API_CLIENT = PROVIDERS["gemini-cli"]?.apiClient;
// === Codex CLI === derive từ registry codex.transport
export const CODEX_CLI_VERSION = PROVIDERS["codex"]?.cliVersion;
// Map Node arch to Gemini CLI arch string (x64/x86/arm64/...)
function geminiCLIArch() {
const a = arch();
@@ -175,6 +178,11 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C
// makes the backend flag the request and answer 429 Quota Exhausted.
export const ANTIGRAVITY_PROMPT_REWRITES = [
{ from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" },
{ from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." },
// Claude Code prepends this line to its system prompt. The Claude-format translator strips it,
// but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions)
// pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED.
{ from: /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/gim, to: "" },
{ from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") }
];
+7 -4
View File
@@ -27,11 +27,14 @@ const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]);
// ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only.
function findModel(models, modelId, aliasOrId) {
if (!models) return undefined;
const found = models.find(m => m.id === modelId);
const baseModelId = typeof modelId === "string"
? modelId.replace(/\([^()]+\)\s*$/, "").trim()
: modelId;
const found = models.find(m => m.id === modelId || m.id === baseModelId);
if (found) return found;
if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined;
const normalized = normalizeModelId(modelId);
if (normalized === modelId) return undefined;
const normalized = normalizeModelId(baseModelId);
if (normalized === baseModelId) return undefined;
return models.find(m => m.id === normalized);
}
@@ -50,7 +53,7 @@ export function findModelName(aliasOrId, modelId) {
}
export function getModelTargetFormat(aliasOrId, modelId) {
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode") && isMuseSparkModel(modelId)) {
if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) {
return FORMATS.OPENAI_RESPONSES;
}
const models = PROVIDER_MODELS[aliasOrId];
+34 -18
View File
@@ -3,10 +3,11 @@ import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js";
import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js";
import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js";
import { cleanJSONSchemaForAntigravity, normalizeGeminiContents } from "../translator/formats/gemini.js";
import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js";
import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js";
// Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63}
function sanitizeFunctionName(name) {
@@ -187,9 +188,12 @@ export class AntigravityExecutor extends BaseExecutor {
};
}
const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" });
const sessionId = toNumericSessionId(rawSessionId) || rawSessionId;
// ─── Standard (non-image) request ───
// Fix contents for Claude models via Antigravity
const contents = body.request?.contents?.map(c => {
const rawContents = (body.request?.contents || []).map(c => {
let role = c.role;
// functionResponse must be role "user" for Claude models
if (c.parts?.some(p => p.functionResponse)) {
@@ -202,21 +206,33 @@ export class AntigravityExecutor extends BaseExecutor {
return true;
});
// Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE)
// don't persist thoughtSignature in their history, so backfill the default signature on any
// functionCall part that arrives without one.
const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false;
if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) {
return {
...c, role,
parts: needsBackfill
? parts.map(p => (p.functionCall && !p.thoughtSignature)
? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE }
: p)
: parts,
};
}
return c;
// don't persist thoughtSignature in their history, so backfill from cache or default signature.
// In parallel function calls, only the first call needs a signature; siblings stay unsigned.
let firstFunctionCallSeen = false;
const modifiedParts = parts?.map(p => {
if (!p.functionCall) return p;
const callId = p.functionCall.id;
const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId, body.model || model) : null;
const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined);
firstFunctionCallSeen = true;
if (callSig) {
return { ...p, thoughtSignature: callSig };
}
if (p.thoughtSignature && !cachedSig) {
// Unsigned sibling call
const { thoughtSignature: _, ...rest } = p;
return rest;
}
return p;
});
return {
...c,
role,
parts: modifiedParts || parts || [],
};
});
const contents = normalizeGeminiContents(rawContents);
// Sanitize tool schemas and function names before sending to Antigravity.
let tools = body.request?.tools;
@@ -267,7 +283,7 @@ export class AntigravityExecutor extends BaseExecutor {
generationConfig,
...(contents && { contents }),
...(tools && { tools }),
sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }),
sessionId,
safetySettings: undefined,
...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } })
};
+1 -1
View File
@@ -127,7 +127,7 @@ export class BaseExecutor {
for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) {
const url = this.buildUrl(model, stream, urlIndex, credentials);
const transformedBody = this.transformRequest(model, body, stream, credentials);
const headers = this.buildHeaders(credentials, stream, url, model);
const headers = this.buildHeaders(credentials, stream, url, model, transformedBody);
if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0;
+11 -1
View File
@@ -12,6 +12,7 @@ import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js";
import { dbg } from "../utils/debugLog.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { stripCodexUnsupportedPatterns } from "../utils/codexToolSchema.js";
// SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts.
const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"];
@@ -72,6 +73,9 @@ function stripStoredItemReferences(body) {
function normalizeCodexTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
// Codex's schema validator has no Unicode property escapes; a `pattern`
// carrying `\p{...}` 400s the whole request on every account (#3922).
const patternStats = { removed: 0 };
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const type = typeof tool.type === "string" ? tool.type : "";
@@ -80,6 +84,9 @@ function normalizeCodexTools(body) {
for (const st of tool.tools) {
const n = typeof st?.name === "string" ? st.name.trim().slice(0, 128) : "";
if (n) validNames.add(n);
if (st?.parameters && typeof st.parameters === "object") {
st.parameters = stripCodexUnsupportedPatterns(st.parameters, patternStats);
}
}
}
return true;
@@ -101,10 +108,13 @@ function normalizeCodexTools(body) {
tool.type = "function";
tool.name = name.slice(0, 128);
if (description) tool.description = description;
tool.parameters = parameters;
tool.parameters = stripCodexUnsupportedPatterns(parameters, patternStats);
validNames.add(name);
return true;
});
if (patternStats.removed > 0) {
dbg("CODEX", `stripped ${patternStats.removed} unsupported tool schema pattern(s)`);
}
// Drop tool_choice if it references an unknown function name
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {
+18 -4
View File
@@ -40,10 +40,24 @@ export class CommandCodeExecutor extends BaseExecutor {
}
async execute(opts) {
const result = await super.execute(opts);
if (!result?.response?.ok || !result.response.body) return result;
result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
return result;
const maxRetries = 2;
for (let attempt = 0; attempt <= maxRetries; attempt++) {
const result = await super.execute(opts);
if (!result?.response?.ok || !result.response.body) return result;
const wrappedResponse = await inspectAndWrapCommandCodeResponse(result.response, opts.model);
if (!wrappedResponse.ok && attempt < maxRetries) {
const isRetryableStatus = wrappedResponse.status === 502 || wrappedResponse.status === 503 || wrappedResponse.status === 504;
if (isRetryableStatus) {
opts.log?.debug?.("RETRY", `CommandCode upstream returned status ${wrappedResponse.status}, retrying ${attempt + 1}/${maxRetries}...`);
await new Promise(r => setTimeout(r, 1000 * (attempt + 1)));
continue;
}
}
result.response = wrappedResponse;
return result;
}
}
parseError(response, bodyText) {
+14 -3
View File
@@ -151,7 +151,7 @@ export class DefaultExecutor extends BaseExecutor {
return BEARER;
}
buildHeaders(credentials, stream = true, url, model) {
buildHeaders(credentials, stream = true, url, model, body = null) {
const rt = credentials?.runtimeTransport;
const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) };
const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor();
@@ -159,8 +159,19 @@ export class DefaultExecutor extends BaseExecutor {
for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials);
applyAuth(headers, desc, credentials);
if (this.provider === "claude" && model) {
headers["Anthropic-Beta"] = selectAnthropicBeta(model);
// anthropic-compatible-* nodes serving a real Claude model sit in front of
// Anthropic itself (a rotating multi-account proxy, a corporate gateway),
// so the request needs the same beta flags the `claude` provider sends:
// without `context-management-2025-06-27` upstream rejects the
// `context_management` block Claude Code puts in every request with
// "context_management: Extra inputs are not permitted" (HTTP 400), and the
// combo silently falls through to the next model. The model id gates this:
// a node fronting Kimi or GLM answers on its own ids and never matches, so
// gateways that would choke on unknown beta flags are left untouched.
const isClaudeModel = typeof model === "string" && /^claude-/.test(model);
if (model && (this.provider === "claude"
|| (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) {
headers["Anthropic-Beta"] = selectAnthropicBeta(model, body);
}
// Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams
+6
View File
@@ -10,12 +10,14 @@ import { CodexExecutor } from "./codex.js";
import { CursorExecutor } from "./cursor.js";
import { VertexExecutor } from "./vertex.js";
import { OpenCodeExecutor } from "./opencode.js";
import { OpenCodeGoExecutor } from "./opencode-go.js";
import { GrokWebExecutor } from "./grok-web.js";
import { GrokCliExecutor } from "./grok-cli.js";
import { PerplexityWebExecutor } from "./perplexity-web.js";
import { OllamaLocalExecutor } from "./ollama-local.js";
import { CommandCodeExecutor } from "./commandcode.js";
import { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
import { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
import { MimoFreeExecutor } from "./mimo-free.js";
import { CodeBuddyExecutor } from "./codebuddy-cn.js";
import { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";
@@ -41,6 +43,7 @@ const executors = {
vertex: new VertexExecutor("vertex"),
"vertex-partner": new VertexExecutor("vertex-partner"),
opencode: new OpenCodeExecutor(),
"opencode-go": new OpenCodeGoExecutor(),
"grok-web": new GrokWebExecutor(),
"grok-cli": new GrokCliExecutor(),
gcli: new GrokCliExecutor(), // Alias
@@ -49,6 +52,7 @@ const executors = {
"ollama-local": new OllamaLocalExecutor(),
commandcode: new CommandCodeExecutor(),
"xiaomi-tokenplan": new XiaomiTokenplanExecutor(),
"xiaomi-mimo": new XiaomiMimoExecutor(),
"mimo-free": new MimoFreeExecutor(),
mmf: new MimoFreeExecutor(), // Alias for mimo-free
"codebuddy-cn": new CodeBuddyExecutor(),
@@ -86,12 +90,14 @@ export { CursorExecutor } from "./cursor.js";
export { VertexExecutor } from "./vertex.js";
export { DefaultExecutor } from "./default.js";
export { OpenCodeExecutor } from "./opencode.js";
export { OpenCodeGoExecutor } from "./opencode-go.js";
export { GrokWebExecutor } from "./grok-web.js";
export { GrokCliExecutor } from "./grok-cli.js";
export { PerplexityWebExecutor } from "./perplexity-web.js";
export { OllamaLocalExecutor } from "./ollama-local.js";
export { CommandCodeExecutor } from "./commandcode.js";
export { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js";
export { XiaomiMimoExecutor } from "./xiaomi-mimo.js";
export { MimoFreeExecutor } from "./mimo-free.js";
export { CodeBuddyExecutor } from "./codebuddy-cn.js";
export { CodeBuddyIntlExecutor } from "./codebuddy-intl.js";
+36 -16
View File
@@ -127,12 +127,18 @@ async function readResponsePrefix(response, signal, maxBytes, timeoutMs) {
return decoder.decode(concatChunks(chunks, totalBytes));
}
// The instruction goes into the current user turn, never into a top-level
// `systemPrompt`: kiro.dev answers any body carrying that field with
// 400 REQUEST_BODY_INVALID, so writing it here turned every repair retry into
// a hard failure.
function appendRepairInstruction(body, kind) {
const repaired = structuredClone(body || {});
const instruction = REPAIR_INSTRUCTIONS[kind] || "Retry the previous incomplete Kiro response.";
repaired.systemPrompt = repaired.systemPrompt
? `${repaired.systemPrompt}\n\n${instruction}`
: instruction;
const msg = repaired?.conversationState?.currentMessage?.userInputMessage;
if (msg) {
const content = typeof msg.content === "string" ? msg.content : "";
msg.content = content ? `${content}\n\n${instruction}` : instruction;
}
return repaired;
}
@@ -259,6 +265,19 @@ export class KiroExecutor extends BaseExecutor {
}
}
// CLIRO parity for the Amazon surfaces: the Kiro runtime accepts the
// SSO bearer header + agent-mode marker. Without these the deprecated
// path gateway answers REQUEST_BODY_INVALID for modern payloads.
if (credentials?.accessToken) {
headers["x-amz-sso-bearer"] = credentials.accessToken;
}
headers["x-amzn-kiro-agent-mode"] = "spec";
headers["x-amzn-codewhisperer-machine-id"] = "kiro-desktop";
const profileArn = credentials?.providerSpecificData?.profileArn;
if (profileArn) {
headers["x-amzn-codewhisperer-profile-arn"] = profileArn;
}
return headers;
}
@@ -285,9 +304,13 @@ export class KiroExecutor extends BaseExecutor {
// 403 "bearer token invalid", so they must hit the CodeWhisperer
// *.amazonaws.com surface, and in the region the token was minted in
// (the baseUrls are hardcoded us-east-1).
const isCodeWhispererSurface =
authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc";
if (!isCodeWhispererSurface) return baseUrls;
// Kiro deprecated the legacy path-style GenerateAssistantResponse on
// runtime.*.kiro.dev (IDE 1.0.228+ moved to POST / + x-amz-target). The
// path gateway now answers valid modern payloads with 400
// REQUEST_BODY_INVALID, and 400 is terminal in BaseExecutor, so kiro.dev
// must never be the first surface for any auth method. Amazon surfaces
// reject foreign tokens with 401/403, which DO fall through, so trying
// q/codewhisperer first is safe for every auth method (CLIRO parity).
const region = (credentials?.providerSpecificData?.region || "us-east-1").trim();
const regionalize = (u) =>
@@ -297,20 +320,17 @@ export class KiroExecutor extends BaseExecutor {
const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize);
const others = baseUrls.filter((u) => !u.includes("amazonaws.com"));
if (authMethod === "api_key") {
const q = amazon.filter((u) => u.includes("://q."));
const remaining = amazon.filter((u) => !u.includes("://q."));
return q.length > 0
? [...q, ...remaining, ...others]
: [...amazon, ...others];
}
return amazon.length > 0 ? [...amazon, ...others] : baseUrls;
const q = amazon.filter((u) => u.includes("://q."));
const remaining = amazon.filter((u) => !u.includes("://q."));
return q.length > 0
? [...q, ...remaining, ...others]
: [...amazon, ...others];
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
const baseUrls = this.getOrderedBaseUrls(credentials);
return baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
const url = baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl;
return url;
}
// Retry only endpoint/auth-surface failures. Payload-invalid HTTP 400 must be
+192
View File
@@ -0,0 +1,192 @@
import crypto from "node:crypto";
import { DefaultExecutor } from "./default.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { modelTargetFormat } from "../providers/models/schema.js";
import { getProviderModels } from "../config/providerModels.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../translator/formats/responsesApi.js";
const SESSION_HEADER = "x-opencode-session";
const SESSION_FIELD = "_opencodeGoSession";
const MAX_SESSION_LENGTH = 256;
const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses";
const MAX_TOOL_NAME_LEN = 128;
function normalizeSession(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return normalized;
}
function nativeSession(headers) {
if (!headers || typeof headers !== "object") return null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value);
}
return null;
}
function translatedSession(sessionId, clientTool) {
const digest = crypto
.createHash("sha256")
.update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`)
.digest("hex")
.slice(0, 32);
return `ses_${digest}`;
}
// Strip the thinking suffix "model(level)" so checks hit the base id.
function baseModelId(model) {
return String(model || "").replace(/\([^()]+\)\s*$/, "").trim();
}
// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …).
// Reading the registry keeps this in sync with config — never hardcode model ids here.
function isResponsesModel(model) {
const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model));
return modelTargetFormat(entry) === "openai-responses";
}
// Flatten Chat Completions tool declarations into the Responses flat shape and
// drop hosted/nameless tools the /responses endpoint rejects.
function normalizeResponsesTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
const name = rawName.trim();
if (!name) return false;
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
? tool.parameters
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
// Mirror the request translator: {type:"object"} without properties is rejected
// by strict Responses backends, so fill in the empty properties map.
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
for (const k of Object.keys(tool)) delete tool[k];
tool.type = "function";
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
if (description) tool.description = description;
tool.parameters = parameters;
validNames.add(tool.name);
return true;
});
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
if (!n || !validNames.has(n)) delete body.tool_choice;
}
}
}
// Last line of defense for native Responses clients (sourceFormat === targetFormat
// skips translation): coerce items in place so malformed tool payloads 400 here
// with a clear shape instead of upstream as InputValidationError.
function sanitizeResponsesItems(body) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
// Strip prior-turn reasoning items: Muse Spark contributor models route to
// an upstream Console backend where encrypted_content cannot be validated across
// rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller".
if (item.type === "reasoning") return false;
delete item.encrypted_content;
delete item.reasoning_encrypted_content;
if (item.type === "function_call") {
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
item.call_id = clampResponsesCallId(item.call_id);
item.arguments = coerceResponsesArguments(item.arguments);
return true;
}
if (item.type === "function_call_output") {
item.call_id = clampResponsesCallId(item.call_id);
item.output = coerceResponsesOutput(item.output);
return true;
}
return true;
});
}
export class OpenCodeGoExecutor extends DefaultExecutor {
constructor() {
super("opencode-go");
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
// Muse Spark lives on /responses even when a stale runtimeTransport leaks in.
if (isResponsesModel(model)) return RESPONSES_BASE_URL;
return super.buildUrl(model, stream, urlIndex, credentials);
}
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
const sourceCredentials = credentials || {};
const native = nativeSession(sourceCredentials.rawHeaders);
const resolved = normalizeSession(providerSessionId) || resolveSessionId({
headers: sourceCredentials.rawHeaders,
body,
connectionId: sourceCredentials.connectionId,
scope: "opencode-go",
});
return {
...sourceCredentials,
[SESSION_FIELD]: native || translatedSession(resolved, clientTool),
};
}
async execute(args) {
const credentials = this.prepareRequestCredentials(args);
return super.execute({ ...args, credentials });
}
buildHeaders(credentials, stream = true, url, model) {
const headers = super.buildHeaders(credentials || {}, stream, url, model);
const prepared = credentials?.[SESSION_FIELD];
if (prepared) {
headers[SESSION_HEADER] = prepared;
return headers;
}
const fallback = this.prepareRequestCredentials({ credentials });
headers[SESSION_HEADER] = fallback[SESSION_FIELD];
return headers;
}
transformRequest(model, body, stream, credentials) {
const out = super.transformRequest(model, body);
if (!isResponsesModel(model || body?.model)) return out;
const normalized = normalizeResponsesInput(out.input);
if (normalized) out.input = normalized;
if (!Array.isArray(out.input) || out.input.length === 0) {
out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
}
// Responses names the output cap max_output_tokens, not max_tokens.
if (out.max_output_tokens === undefined) {
if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens;
else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens;
}
delete out.max_tokens;
delete out.max_completion_tokens;
if (out.reasoning_effort !== undefined && out.reasoning === undefined) {
out.reasoning = { effort: out.reasoning_effort, summary: "auto" };
}
if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) {
if (!out.reasoning.summary) out.reasoning.summary = "auto";
}
delete out.reasoning_effort;
out.stream = true;
out.store = false;
normalizeResponsesTools(out);
sanitizeResponsesItems(out);
return out;
}
}
+456 -25
View File
@@ -1,24 +1,311 @@
import crypto from "crypto";
import { BaseExecutor } from "./base.js";
import { PROVIDERS } from "../config/providers.js";
import { MEMORY_CONFIG } from "../config/runtimeConfig.js";
import { getThinkingLevels } from "../providers/thinkingLevels.js";
import { injectReasoningContent } from "../utils/reasoningContentInjector.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { isMuseSparkModel } from "../providers/models/helpers.js";
import { ANTHROPIC_API_VERSION } from "../providers/shared.js";
import {
normalizeResponsesInput,
clampResponsesCallId,
coerceResponsesArguments,
coerceResponsesOutput,
} from "../translator/formats/responsesApi.js";
const OPENCODE_UA = "opencode";
const OPENCODE_UA = "opencode/1.18.31";
const MAX_SESSION_LENGTH = 256;
const MAX_TOOL_NAME_LEN = 128;
const SESSION_HEADER = "x-opencode-session";
const SESSION_FIELD = "_opencodeSession";
const REQ_FIELD = "_opencodeRequest";
export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/;
const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
// OpenCode free tier requires both 'bash' and 'read' in tools payload.
// Injected as cloaked decoy tools so external CLI tools (e.g. Claude Code's Bash/Read)
// take precedence while satisfying upstream verification.
const OPENCODE_DECOY_CHAT_TOOLS = [
{
type: "function",
function: {
name: "bash",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
},
{
type: "function",
function: {
name: "read",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
},
];
const OPENCODE_DECOY_RESPONSES_TOOLS = [
{
type: "function",
name: "bash",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
{
type: "function",
name: "read",
description: "This tool is currently unavailable and must not be used.",
parameters: { type: "object", properties: {} },
},
];
function cloakOpencodeTools(body, isResponses) {
if (!body || typeof body !== "object") return;
if (isResponses) {
if (!Array.isArray(body.tools)) body.tools = [];
const names = new Set(body.tools.map((t) => t.name || t.function?.name));
for (const tool of OPENCODE_DECOY_RESPONSES_TOOLS) {
if (!names.has(tool.name)) body.tools.push({ ...tool });
}
if (!body.tool_choice) body.tool_choice = "auto";
} else {
const hasTools = Array.isArray(body.tools) && body.tools.length > 0;
if (!hasTools) {
body.tools = OPENCODE_DECOY_CHAT_TOOLS.map((t) => ({ ...t, function: { ...t.function } }));
if (!body.tool_choice) body.tool_choice = "none";
} else {
const names = new Set(body.tools.map((t) => t.function?.name || t.name));
for (const tool of OPENCODE_DECOY_CHAT_TOOLS) {
if (!names.has(tool.function.name)) {
body.tools.push({ ...tool, function: { ...tool.function } });
}
}
}
}
}
function hasValidOpencodeVersion(ua) {
const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i);
if (!m) return false;
const major = parseInt(m[1], 10);
const minor = parseInt(m[2], 10);
return major > 1 || (major === 1 && minor >= 17);
}
// Models served by /zen/v1/responses; every other model stays on /chat/completions.
const RESPONSES_MODELS = new Set([
"muse-spark-1.2-contributor-free",
"muse-spark-1.3-contributor-free",
]);
const MESSAGES_MODELS = new Set(["union-alpha"]);
function generateRequestId() {
return `msg_${crypto.randomUUID().replace(/-/g, "")}`;
let lastTimestamp = 0;
let counter = 0;
function unstableRandom() {
const bytes = crypto.randomBytes(14);
let randomPart = "";
for (let i = 0; i < 14; i++) {
randomPart += BASE62_CHARS[bytes[i] % 62];
}
return randomPart;
}
function generateSessionId() {
return `ses_${crypto.randomUUID().replace(/-/g, "")}`;
export function generateSessionId(timestamp = Date.now()) {
if (timestamp !== lastTimestamp) {
lastTimestamp = timestamp;
counter = 0;
}
counter++;
const current = BigInt(timestamp) * 0x1000n + BigInt(counter);
const value = ~current;
const time = Array.from({ length: 6 }, (_, index) =>
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
.toString(16)
.padStart(2, "0")
).join("");
return `ses_${time}${unstableRandom()}`;
}
export function generateRequestId(timestamp = Date.now()) {
const current = BigInt(timestamp) * 0x1000n + 1n;
const value = current;
const time = Array.from({ length: 6 }, (_, index) =>
Number((value >> BigInt(40 - 8 * index)) & 0xffn)
.toString(16)
.padStart(2, "0")
).join("");
return `msg_${time}${unstableRandom()}`;
}
export function translateSessionId(sessionId, clientTool = "") {
if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) {
return sessionId.trim();
}
const digest = crypto
.createHash("sha256")
.update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`)
.digest();
const timeHex = digest.subarray(0, 6).toString("hex");
let randomPart = "";
for (let i = 6; i < 20; i++) {
randomPart += BASE62_CHARS[digest[i] % 62];
}
return `ses_${timeHex}${randomPart}`;
}
function normalizeSession(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return normalized;
}
function nativeSession(headers) {
if (!headers || typeof headers !== "object") return null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) {
const normalized = normalizeSession(value);
if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized;
}
}
return null;
}
// Upstream free-tier quota is accounted per session. Minting a fresh
// x-opencode-session on every request burns through it and surfaces as
// 429 FreeUsageLimitError with growing reset-after delays, while the real
// CLI reuses one long-lived canonical session per conversation. Mirror
// that: one stable canonical session per downstream identity, evicted
// after MEMORY_CONFIG.sessionTtlMs like the other session stores.
const stableOpencodeSessions = new Map();
const MAX_STABLE_SESSIONS = 1000;
const stableSessionCleanup = setInterval(() => {
const now = Date.now();
for (const [key, entry] of stableOpencodeSessions) {
if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) {
stableOpencodeSessions.delete(key);
}
}
}, MEMORY_CONFIG.sessionCleanupIntervalMs);
if (stableSessionCleanup.unref) stableSessionCleanup.unref();
function identityKey(credentials) {
const connectionId = credentials?.connectionId || credentials?.id;
if (connectionId) return `opencode:conn:${String(connectionId).slice(0, 128)}`;
const raw = credentials?.rawHeaders || {};
const auth = raw.authorization || raw.Authorization || raw["x-api-key"] || raw["X-Api-Key"] || "";
if (auth) {
const digest = crypto.createHash("sha256").update(String(auth)).digest("hex").slice(0, 32);
return `opencode:auth:${digest}`;
}
return "opencode:default";
}
export function stableSessionId(credentials) {
const key = identityKey(credentials);
const existing = stableOpencodeSessions.get(key);
if (existing) {
existing.lastUsed = Date.now();
stableOpencodeSessions.delete(key);
stableOpencodeSessions.set(key, existing);
return existing.sessionId;
}
const sessionId = generateSessionId();
if (stableOpencodeSessions.size >= MAX_STABLE_SESSIONS) {
stableOpencodeSessions.delete(stableOpencodeSessions.keys().next().value);
}
stableOpencodeSessions.set(key, { sessionId, lastUsed: Date.now() });
return sessionId;
}
function lastUserText(body) {
try {
if (!body || typeof body !== "object") return "";
const arr = Array.isArray(body.messages)
? body.messages
: Array.isArray(body.input)
? body.input
: null;
if (!arr) return typeof body.input === "string" ? body.input.slice(-600) : "";
for (let i = arr.length - 1; i >= 0; i--) {
const msg = arr[i];
if (!msg) continue;
if (msg.role && msg.role !== "user") continue;
const content = msg.content;
if (typeof content === "string" && content.trim()) return content.trim().slice(-600);
if (Array.isArray(content)) {
const text = content
.map((part) => (typeof part === "string" ? part : part?.text || part?.input_text || ""))
.join(" ")
.trim();
if (text) return text.slice(-600);
}
}
} catch {
return "";
}
return "";
}
// The real CLI sends the current user message id (stable per turn, same on
// retries) as x-opencode-request. Derive it deterministically from the
// session plus the last user message so retries share the id.
export function deriveRequestId(sessionId, body) {
const text = lastUserText(body);
if (!text) return generateRequestId();
const digest = crypto
.createHash("sha256")
.update(`opencode-req\0${sessionId || ""}\0${text}`)
.digest();
const timeHex = digest.subarray(0, 6).toString("hex");
let randomPart = "";
for (let i = 6; i < 20; i++) {
randomPart += BASE62_CHARS[digest[i] % 62];
}
const id = `msg_${timeHex}${randomPart}`;
return OPENCODE_REQUEST_RE.test(id) ? id : generateRequestId();
}
function normalizeRequestId(value) {
if (typeof value !== "string") return null;
const normalized = value.trim();
if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null;
return OPENCODE_REQUEST_RE.test(normalized) ? normalized : null;
}
function bodyHasSessionHints(body) {
try {
if (!body || typeof body !== "object") return false;
if (typeof body.session_id === "string" && body.session_id.trim()) return true;
if (typeof body.conversation_id === "string" && body.conversation_id.trim()) return true;
if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key.trim()) return true;
if (body.metadata && typeof body.metadata.user_id === "string" && body.metadata.user_id.trim()) return true;
if (body.request && body.request.sessionId != null && String(body.request.sessionId) !== "") return true;
const arr = Array.isArray(body.messages)
? body.messages
: Array.isArray(body.input)
? body.input
: null;
if (arr) {
let assistantText = "";
for (const msg of arr) {
if (msg?.role === "assistant") {
const content = msg.content;
if (typeof content === "string") assistantText += content;
else if (Array.isArray(content)) {
for (const part of content) assistantText += part?.text || part?.output || "";
}
if (assistantText.length >= 50) return true;
}
}
}
return false;
} catch {
return false;
}
}
// Strip the thinking suffix "model(level)" so registry lookups hit the base id.
@@ -31,14 +318,116 @@ function isResponsesModel(model) {
return RESPONSES_MODELS.has(base) || isMuseSparkModel(base);
}
function resolveOpencodeSession(body, credentials) {
function isMessagesModel(model) {
return MESSAGES_MODELS.has(baseModelId(model));
}
function resolveOpencodeSession(body, credentials, providerSessionId, clientTool) {
const headers = credentials?.rawHeaders || {};
return resolveSessionId({
headers,
body,
connectionId: credentials?.connectionId,
scope: "opencode",
generate: generateSessionId,
const native = nativeSession(headers);
if (native) return native;
let incoming = null;
for (const [key, value] of Object.entries(headers)) {
if (key.toLowerCase() === SESSION_HEADER) {
incoming = normalizeSession(value);
break;
}
}
const hinted = incoming || normalizeSession(providerSessionId);
if (hinted) return translateSessionId(hinted, clientTool);
if (credentials?.connectionId || bodyHasSessionHints(body)) {
let viaManager = null;
try {
viaManager = resolveSessionId({
headers,
body,
connectionId: credentials?.connectionId,
scope: "opencode",
});
} catch {
viaManager = null;
}
if (viaManager) return translateSessionId(viaManager, clientTool);
}
return stableSessionId(credentials);
}
function resolveOpencodeRequestId(body, credentials, sessionId) {
const raw = credentials?.rawHeaders || {};
for (const [key, value] of Object.entries(raw)) {
if (key.toLowerCase() === "x-opencode-request") {
const normalized = normalizeRequestId(value);
if (normalized) return normalized;
break;
}
}
return deriveRequestId(sessionId, body);
}
function normalizeResponsesTools(body) {
if (!Array.isArray(body.tools)) return;
const validNames = new Set();
body.tools = body.tools.filter((tool) => {
if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false;
const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null;
const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : "");
const name = rawName.trim();
if (!name) return false;
const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : "");
let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters))
? tool.parameters
: (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} });
if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} };
for (const k of Object.keys(tool)) delete tool[k];
tool.type = "function";
tool.name = name.slice(0, MAX_TOOL_NAME_LEN);
if (description) tool.description = description;
tool.parameters = parameters;
validNames.add(tool.name);
return true;
});
if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) {
if (body.tool_choice.type === "function") {
const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : "";
if (!n || !validNames.has(n)) delete body.tool_choice;
}
}
}
function sanitizeResponsesItems(body) {
if (!Array.isArray(body.input)) return;
body.input = body.input.filter((item) => {
if (!item || typeof item !== "object" || Array.isArray(item)) return true;
// Strip prior-turn reasoning items: OpenCode Free uses public/pooled credentials
// (`Bearer public`) routing to an upstream OpenAI/Console account pool.
// OpenAI Responses API strictly enforces that reasoning `encrypted_content`
// can only be decrypted by the exact caller/account that issued it; sending it
// across different accounts or rotating proxy relays triggers:
// [invalid_request_error] reasoning `encrypted_content` was not issued to this caller (400).
// Furthermore, under stateless mode (store=false), omitting encrypted_content
// causes OpenAI to reject the referenced reasoning item as "not found or was deleted".
// Dropping prior reasoning items allows multi-turn conversations and tool-calling
// loops to succeed cleanly.
if (item.type === "reasoning") return false;
delete item.encrypted_content;
delete item.reasoning_encrypted_content;
if (item.type === "function_call") {
if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false;
item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN);
item.call_id = clampResponsesCallId(item.call_id);
item.arguments = coerceResponsesArguments(item.arguments);
return true;
}
if (item.type === "function_call_output") {
item.call_id = clampResponsesCallId(item.call_id);
item.output = coerceResponsesOutput(item.output);
return true;
}
return true;
});
}
@@ -74,12 +463,35 @@ const IP_LIMIT_BODY = /limit|rate|quota|exhausted|capacity|too many|retry/i;
export class OpenCodeExecutor extends BaseExecutor {
constructor() {
super("opencode", PROVIDERS.opencode);
this._currentSessionId = null;
}
prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) {
const sourceCredentials = credentials || {};
const session = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool);
return {
...sourceCredentials,
[SESSION_FIELD]: session,
[REQ_FIELD]: resolveOpencodeRequestId(body, sourceCredentials, session),
};
}
transformRequest(model, body, stream, credentials) {
this._currentSessionId = resolveOpencodeSession(body, credentials);
if (isResponsesModel(model)) {
if (body && typeof body === "object" && model && !body.model) body.model = model;
// Zen rejects non-streaming requests on free models with 403 FreeTierError;
// always stream upstream and let the handler layer aggregate for non-stream clients.
if (body && typeof body === "object") body.stream = true;
if (isResponsesModel(model || body?.model) && body && typeof body === "object") {
// ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng.
if ("tool_choice" in body && body.tool_choice !== "auto"
&& this.config.quirks?.forceAutoToolChoiceModels?.includes(baseModelId(model))) {
body.tool_choice = "auto";
}
const normalized = normalizeResponsesInput(body.input);
if (normalized) body.input = normalized;
if (!Array.isArray(body.input) || body.input.length === 0) {
body.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }];
}
// Responses API names the output cap max_output_tokens and takes thinking
// as reasoning:{effort,summary} — normalize the Chat fields at this boundary.
if (body.max_output_tokens === undefined) {
@@ -89,35 +501,54 @@ export class OpenCodeExecutor extends BaseExecutor {
delete body.max_tokens;
delete body.max_completion_tokens;
normalizeOpencodeReasoning(model, body);
body.stream = true;
body.store = false;
normalizeResponsesTools(body);
sanitizeResponsesItems(body);
if (!Array.isArray(body.tools) || body.tools.length === 0) {
cloakOpencodeTools(body, true);
}
} else if (body && typeof body === "object") {
cloakOpencodeTools(body, false);
}
return injectReasoningContent({ provider: this.provider, model, body });
}
buildUrl(model) {
const base = this.config.baseUrl;
return isResponsesModel(model)
? `${base}/zen/v1/responses`
: `${base}/zen/v1/chat/completions`;
async execute(args) {
return super.execute({ ...args, credentials: this.prepareRequestCredentials(args) });
}
buildHeaders(credentials, stream = true) {
buildUrl(model) {
const base = this.config.baseUrl;
if (isResponsesModel(model)) return `${base}/zen/v1/responses`;
if (isMessagesModel(model)) return `${base}/zen/v1/messages`;
return `${base}/zen/v1/chat/completions`;
}
buildHeaders(credentials, stream = true, url = "") {
const raw = credentials?.rawHeaders || {};
const lower = {};
for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v;
const downstreamUa = lower["user-agent"] || "";
const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode");
const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa);
return {
const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD];
const downstreamReq = normalizeRequestId(lower["x-opencode-request"]);
const requestId = credentials?.[REQ_FIELD] || downstreamReq || generateRequestId();
const headers = {
"Content-Type": "application/json",
"Authorization": "Bearer public",
"User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA,
"x-opencode-client": lower["x-opencode-client"] || "desktop",
"x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(),
"x-opencode-request": lower["x-opencode-request"] || generateRequestId(),
"x-opencode-session": session,
"x-opencode-request": requestId,
"x-opencode-project": lower["x-opencode-project"] || "global",
"Accept": stream ? "text/event-stream" : "*/*",
};
if (url.endsWith("/messages")) headers["anthropic-version"] = ANTHROPIC_API_VERSION;
return headers;
}
parseError(response, bodyText) {
+150 -31
View File
@@ -31,16 +31,21 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { SSE_DONE } from "../utils/sseConstants.js";
import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js";
import {
QODER_CHAT_URL_ENCODED,
QODER_CHAT_BASE_ALT,
QODER_CHAT_SIG_PATH,
QODER_MODEL_MAP,
QODER_CONTEXT_TIER_ENV,
qoderInferenceBase,
} from "../shared/qoder/constants.js";
import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js";
import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js";
import { encodeDataUri } from "../translator/concerns/image.js";
import { createQoderSseCoalescer } from "../shared/qoder/sse.js";
import { rewriteQoderMessageAttachments } from "../shared/qoder/attachments.js";
import { resolveQoderContextTier, applyQoderContextTier } from "../shared/qoder/contextTier.js";
/**
* Hoist role:"system" messages out of the messages array (Qoder rejects
* system in messages) and flatten any multipart content arrays.
* system in messages) and flatten multipart content arrays — EXCEPT image
* blocks, which are preserved (see normalizeContent).
*/
function normalizeMessages(messages) {
if (!Array.isArray(messages) || messages.length === 0) {
@@ -50,18 +55,88 @@ function normalizeMessages(messages) {
const out = [];
for (const msg of messages) {
if (!msg || typeof msg !== "object") continue;
const text = extractText(msg.content);
if (msg.role === "system") {
const text = extractText(msg.content);
if (text) systemParts.push(text);
continue;
}
const cloned = { ...msg };
cloned.content = text;
cloned.content = normalizeContent(msg.content);
out.push(cloned);
}
return { messages: out, systemText: systemParts.join("\n\n") };
}
/**
* Normalize one message's content for Qoder.
*
* Text-only content is flattened to a plain string (Qoder's historical
* shape). When images are present the content stays an array and image
* blocks are kept as OpenAI-style `image_url` parts. Native qodercli
* uploads inlined bytes to `/api/v2/image/upload` first and then sends
* the OSS URL — `buildQoderRequestBody` does that rewrite before this
* runs. Tiny leftover data URIs are still accepted. The legacy
* top-level `image_urls` / `chat_context.imageUrls` slots stay null —
* qodercli leaves them null too.
*
* Claude-style `{type:"image", source:{...}}` blocks are converted to
* `image_url`. File/document blocks that survived rewrite become short
* stubs so 30MB PDFs never land in agent_chat_generation.
*/
function normalizeContent(content) {
if (typeof content === "string") return content;
if (content == null) return "";
if (!Array.isArray(content)) return String(content);
const blocks = [];
const textParts = [];
let hasImage = false;
const pushText = (text) => {
if (!text) return;
if (hasImage || blocks.length) blocks.push({ type: OPENAI_BLOCK.TEXT, text });
else textParts.push(text);
};
const imageUrlOf = (item) => {
if (typeof item.image_url === "string" && item.image_url) return item.image_url;
if (typeof item.image_url?.url === "string" && item.image_url.url) return item.image_url.url;
return null;
};
for (const item of content) {
if (!item || typeof item !== "object") continue;
const imageUrl = item.type === OPENAI_BLOCK.IMAGE_URL ? imageUrlOf(item) : null;
if (imageUrl) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: imageUrl } });
hasImage = true;
} else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) {
// Claude base64/url image → OpenAI image_url equivalent.
const src = item.source;
const url = src.type === "base64" && src.data
? encodeDataUri(src.media_type || "image/png", src.data)
: typeof src.url === "string" && src.url ? src.url : null;
if (url) {
blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } });
hasImage = true;
}
} else if (item.type === OPENAI_BLOCK.FILE) {
const name = item.file?.filename || item.file?.name || "file";
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
} else if (item.type === CLAUDE_BLOCK.DOCUMENT) {
const name = item.title || "document";
pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`);
} else if (typeof item.text === "string" && item.text) {
pushText(item.text);
}
}
if (!hasImage) return textParts.join("\n");
// Prepend any text collected before the first image block.
if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") });
return blocks;
}
function extractText(content) {
if (typeof content === "string") return content;
if (content == null) return "";
@@ -84,9 +159,9 @@ function extractText(content) {
function lastUserText(messages) {
for (let i = messages.length - 1; i >= 0; i--) {
const m = messages[i];
if (m?.role === "user" && typeof m.content === "string") {
return m.content;
}
if (m?.role !== "user") continue;
if (typeof m.content === "string") return m.content;
if (Array.isArray(m.content)) return extractText(m.content);
}
return "";
}
@@ -110,6 +185,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) {
if (m.role) { h.update("\0"); h.update(m.role); }
if (typeof m.content === "string" && m.content) {
h.update("\0"); h.update(m.content);
} else if (Array.isArray(m.content)) {
// Include image refs so the same prompt with a different image gets
// a distinct chat_record_id.
h.update("\0");
try { h.update(JSON.stringify(m.content)); } catch {}
}
}
if (tools) {
@@ -127,7 +207,7 @@ function truncate(s, n) {
/**
* Map the OpenAI-style request body into the exact shape Qoder expects.
*/
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }) {
async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null }) {
const qoderKey = String(model || "").replace(/^qoder\//, "");
// Fetch model config from dynamic API instead of relying on static QODER_MODEL_MAP.
@@ -146,7 +226,30 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
modelConfig = { ...retried, key: qoderKey };
}
const { messages, systemText } = normalizeMessages(body.messages || []);
const incoming = Array.isArray(body.messages)
? body.messages.map((m) => {
if (!m || typeof m !== "object") return m;
return {
...m,
content: Array.isArray(m.content)
? m.content.map((b) => (b && typeof b === "object" ? { ...b } : b))
: m.content,
};
})
: [];
try {
await rewriteQoderMessageAttachments(incoming, {
credentials,
log,
proxyOptions,
signal,
uploadFn,
});
} catch (err) {
log?.warn?.("QODER", `attachment rewrite failed: ${err.message}`);
}
const { messages, systemText } = normalizeMessages(incoming);
const tools = body.tools;
const isReasoning = !!modelConfig.is_reasoning;
const maxOutputTokens = Number(modelConfig.max_output_tokens) || 0;
@@ -165,7 +268,21 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
const sessionId = stableHash("qoder-session", psd.userId, qoderKey);
const recordId = stableChatRecordId(qoderKey, messages, tools, maxTokens);
return {
// Context-window tier (200K/400K/1M): the IDE picks one from model_config.context_config;
// qodercli-style requests default to the smallest. Escalate when the prompt no longer fits.
const tierChoice = resolveQoderContextTier(
modelConfig,
{ system: systemText, messages, tools },
{ preference: process.env[QODER_CONTEXT_TIER_ENV] },
);
if (tierChoice) {
log?.info?.(
"QODER",
`context tier ${tierChoice.tier.name} (${tierChoice.tier.tokenCount} tokens, ${tierChoice.reason}) for ~${tierChoice.estimatedTokens} prompt tokens`,
);
}
const built = {
qoderKey,
payload: {
request_id: uuidv4(),
@@ -213,6 +330,8 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio
},
modelConfig,
};
if (tierChoice) applyQoderContextTier(built.payload, tierChoice.tier);
return built;
}
/**
@@ -276,6 +395,11 @@ async function peekFirstQoderFrame(reader, decoder) {
* response.text() which hangs until the socket closes — so on terminal
* events we cancel the upstream reader and close our stream immediately.
*
* Usage: Qoder puts finish_reason on `delta` and sends token counts on a
* later `choices: []` frame. Downstream OpenAI/Claude clients only read
* usage from the finish chunk, so we coalesce those two frames (see
* createQoderSseCoalescer) before forwarding.
*
* NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl).
* If detected, return 403 response so chatCore marks connection unavailable
* and triggers combo fallback instead of leaking error text into chat.
@@ -302,6 +426,11 @@ async function wrapQoderSSE(response, model) {
const upstreamDrained = peek.upstreamDone === true;
const encoder = new TextEncoder();
let doneEmitted = false;
const coalescer = createQoderSseCoalescer({ model, encoder, sseDone: SSE_DONE });
const syncDone = () => {
if (coalescer.doneEmitted) doneEmitted = true;
};
// Process one already-extracted SSE line (no trailing newline).
const processLine = (line, controller) => {
@@ -312,15 +441,17 @@ async function wrapQoderSSE(response, model) {
const data = trimmed.slice(5).trimStart();
if (data === "[DONE]") {
controller.enqueue(encoder.encode(SSE_DONE));
doneEmitted = true;
coalescer.flush(controller);
syncDone();
return;
}
let envelope;
try { envelope = JSON.parse(data); } catch { return; }
const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200;
const inner = typeof envelope.body === "string" ? envelope.body : "";
const inner = typeof envelope.body === "string"
? envelope.body
: envelope.body != null ? JSON.stringify(envelope.body) : "";
if (statusVal !== 200) {
const msg = inner || `upstream status ${statusVal}`;
const errChunk = JSON.stringify({
@@ -336,14 +467,8 @@ async function wrapQoderSSE(response, model) {
return;
}
if (!inner) return;
if (inner === "[DONE]") {
controller.enqueue(encoder.encode(SSE_DONE));
doneEmitted = true;
return;
}
// Strip embedded newlines so the SSE frame stays a single event.
const sanitized = inner.replace(/\r?\n/g, "");
controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`));
coalescer.handleInner(inner, controller);
syncDone();
};
const stream = new ReadableStream({
@@ -402,7 +527,7 @@ async function wrapQoderSSE(response, model) {
} finally {
if (!doneEmitted) {
try {
controller.enqueue(encoder.encode(SSE_DONE));
coalescer.flush(controller);
doneEmitted = true;
} catch { /* already closed */ }
}
@@ -431,13 +556,7 @@ export class QoderExecutor extends BaseExecutor {
}
buildUrl(credentials) {
// Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt-
// with "Login expired" (403). Device tokens (dt-...) stay on api3.
const raw = credentials?.apiKey || credentials?.accessToken;
if (typeof raw === "string" && !raw.startsWith("pt-") && (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))) {
return `${QODER_CHAT_BASE_ALT}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
}
return QODER_CHAT_URL_ENCODED;
return `${qoderInferenceBase(credentials)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`;
}
// Override execute entirely — Qoder needs:
+99
View File
@@ -0,0 +1,99 @@
import { DefaultExecutor } from "./default.js";
import { getMimoAccountCookie, invalidateMimoAccountCookieCache, MIMO_API_BASE, MIMO_API_UA } from "../shared/mimoAccount.js";
// Desktop-exclusive Preview models. These are served by the account service's
// /api/route proxy, authorized by the Xiaomi account session (NOT the sk- key).
// See shared/mimoAccount.js for the session handshake.
const PREVIEW_MODELS = new Set(["mimo-x-pro-preview", "mimo-x-flash-preview"]);
// Session cookie resolved in execute() (async) and read back by buildHeaders()
// (sync — BaseExecutor.execute does not await it). Carried on the per-request
// credentials object, same as runtimeTransport.
const COOKIE_KEY = "__mimoAccountCookie";
// Upstream calls may hand us either the bare id or a `provider/model` ref.
function bareModel(model) {
const s = String(model || "");
const i = s.indexOf("/");
return i >= 0 ? s.slice(i + 1) : s;
}
export class XiaomiMimoExecutor extends DefaultExecutor {
constructor() {
super("xiaomi-mimo");
}
static isPreviewModel(model) {
return PREVIEW_MODELS.has(bareModel(model));
}
buildUrl(model, stream, urlIndex = 0, credentials = null) {
// Preview models live on the account-service route, which is not one of the
// declared transports — resolve it before the default runtimeTransport path.
if (XiaomiMimoExecutor.isPreviewModel(model)) {
return `${MIMO_API_BASE}/api/route/chat/completions`;
}
// Cloud API models keep default handling, so a Claude-format client reaches
// the /anthropic/v1/messages transport.
return super.buildUrl(model, stream, urlIndex, credentials);
}
buildHeaders(credentials, stream = true, url, model) {
if (XiaomiMimoExecutor.isPreviewModel(model) && credentials?.[COOKIE_KEY]) {
// Preview models authenticate with the account-session cookie, not the key.
return {
"Content-Type": "application/json",
Accept: stream ? "text/event-stream" : "application/json",
"User-Agent": MIMO_API_UA,
Cookie: credentials[COOKIE_KEY],
};
}
return super.buildHeaders(credentials, stream, url, model);
}
transformRequest(model, body, stream, credentials) {
// super runs stripUnsupportedParams, which flattens Preview content-part
// arrays (see the xiaomi-mimo rule in translator/concerns/paramSupport.js).
const out = super.transformRequest(model, body, stream, credentials);
// Preview models: thinking/params get defaults only — never override what the
// caller set explicitly. (body.model is already `xiaomi/<id>` via upstreamModelId.)
if (XiaomiMimoExecutor.isPreviewModel(model)) {
if (out.thinking == null) out.thinking = { type: "enabled" };
if (out.temperature == null) out.temperature = 1.0;
if (out.top_p == null) out.top_p = 0.95;
if (!out.max_tokens) out.max_tokens = 4096;
}
return out;
}
async execute(args) {
const { model, credentials, proxyOptions = null } = args;
if (!XiaomiMimoExecutor.isPreviewModel(model)) return super.execute(args);
const cookie = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions);
if (!cookie) {
throw new Error(
"Xiaomi MiMo account session unavailable. Sign in to MiMo Desktop once so its passToken is present, then retry.",
);
}
credentials[COOKIE_KEY] = cookie;
const result = await super.execute(args);
// A cached session can expire early — drop it and retry once with a fresh one.
if (result.response.status === 401) {
invalidateMimoAccountCookieCache();
const fresh = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions).catch(() => null);
if (fresh) {
credentials[COOKIE_KEY] = fresh;
return super.execute(args);
}
}
return result;
}
}
export const __test__ = { PREVIEW_MODELS, bareModel, COOKIE_KEY };
export default XiaomiMimoExecutor;
+19 -5
View File
@@ -29,11 +29,18 @@ import {
zedLlmFetch,
} from "../shared/zedAuth.js";
// Wire values for the `provider` field of POST /completions. These are NOT
// display names: cloud.zed.dev matches them exactly, and an unrecognized value
// fails the whole request with `500 {"message":"An internal server error
// occurred."}` before the model is ever looked at. Spellings come from Zed's
// own GET /models catalog: `anthropic`, `open_ai`, `google` (note underscore),
// `x_ai` follows the same convention — so feeding a catalog value back through
// normalizeZedProvider is identity.
const ZED_PROVIDER = {
anthropic: "Anthropic",
openai: "OpenAi",
google: "Google",
xai: "XAi",
anthropic: "anthropic",
openai: "open_ai",
google: "google",
xai: "x_ai",
};
function normalizeZedProvider(value, model) {
@@ -55,7 +62,14 @@ function buildProviderRequest(provider, model, body, stream, credentials) {
return openaiToClaudeRequest(model, body, true);
}
if (provider === ZED_PROVIDER.google) {
return openaiToGeminiRequest(model, body, true);
const geminiRequest = openaiToGeminiRequest(model, body, true);
// Zed's hosted Gemini backend speaks the Vertex safety vocabulary, not the
// public Gemini API enum the shared translator emits (`OFF`, `CIVIC_INTEGRITY`,
// `DANGEROUS_CONTENT`). Drop client-side safetySettings for the Zed Google
// path so Zed applies its own defaults — scoped here so native Gemini/
// Antigravity is untouched.
delete geminiRequest.safetySettings;
return geminiRequest;
}
if (provider === ZED_PROVIDER.openai) {
return openaiToOpenAIResponsesRequest(model, body, true, credentials);
+19 -5
View File
@@ -28,7 +28,7 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js";
import { getCapabilitiesForModel } from "../providers/capabilities.js";
import { stripUnsupportedModalities } from "../translator/concerns/modality.js";
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js";
import { defaultClaudeToolType } from "../translator/concerns/toolCall.js";
import { defaultClaudeToolType, shouldDefaultClaudeToolType } from "../translator/concerns/toolCall.js";
import { resolveSessionId } from "../utils/sessionManager.js";
import { markPoolUnfit, clearPoolUnfit } from "../services/proxyPoolFitness.js";
@@ -249,7 +249,11 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax)
// reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing.
if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) {
// Provider-scoped via quirks (shouldDefaultClaudeToolType): only gateways that declare
// requireClaudeToolType get the explicit type. Applying it unconditionally breaks
// Claude-format endpoints that only accept the legacy typeless tool shape — DeepSeek's
// Anthropic-compatible endpoint 400s with "unknown variant `custom`" (#3905).
if (shouldDefaultClaudeToolType(provider, finalFormat, translatedBody.tools, PROVIDERS)) {
translatedBody.tools = defaultClaudeToolType(translatedBody.tools);
}
@@ -416,7 +420,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
const executeWithPoolFallback = async (attempt = 0) => {
let result;
try {
result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
result = await executor.execute({ model, body: translatedBody, stream, credentials, providerSessionId: sessionSeed, clientTool, signal: streamController.signal, log, proxyOptions });
} catch (error) {
if (error?.poolScoped && typeof resolveProxyConfig === "function" && attempt < MAX_POOL_RETRIES) {
if (await tryNextPool(error.poolScoped, error.message)) return executeWithPoolFallback(attempt + 1);
@@ -497,7 +501,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); }
}
try {
const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions });
const retryResult = await executor.execute({
model,
body: translatedBody,
stream,
credentials,
providerSessionId: sessionSeed,
clientTool,
signal: streamController.signal,
log,
proxyOptions,
});
if (retryResult.response.ok) {
providerResponse = retryResult.response;
providerUrl = retryResult.url;
@@ -556,7 +570,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred
// Streaming response
const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx });
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId });
return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, credentials });
}
export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) {
@@ -6,6 +6,7 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin
import { createErrorResult } from "../../utils/error.js";
import { HTTP_STATUS } from "../../config/runtimeConfig.js";
import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js";
import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js";
import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js";
import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js";
import { decloakToolNames } from "../../utils/claudeCloaking.js";
@@ -319,6 +320,11 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m
}
}
// Unwrap before any consumer reads choices/usage so non-stream clients get a
// bare OpenAI body and usage tracking sees data.usage. No-op unless the
// provider opts in via transport.quirks.clineEnvelope.
responseBody = unwrapClineEnvelope(responseBody, provider);
reqLogger.logProviderResponse(providerResponse.status, providerResponse.statusText, providerResponse.headers, responseBody);
// Unwrap AFTER logging (raw envelope stays in the log for forensics) but
// BEFORE usage extraction/translation so choices/usage resolve downstream.
+14 -8
View File
@@ -3,8 +3,9 @@ import { needsTranslation } from "../../translator/index.js";
import { createSSETransformStreamWithLogger, createPassthroughStreamWithLogger } from "../../utils/stream.js";
import { pipeWithDisconnect } from "../../utils/streamHandler.js";
import { PROVIDERS } from "../../config/providers.js";
import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
import { HTTP_STATUS, STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js";
import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js";
import { buildStreamErrorBytes } from "../../utils/streamHelpers.js";
import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js";
import { saveRequestDetail } from "@/lib/usageDb.js";
import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js";
@@ -22,7 +23,7 @@ const CODEX_SOURCE_TO_TARGET = {
/**
* Determine which SSE transform stream to use based on provider/format.
*/
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) {
function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) {
const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli");
// Responses-API providers (e.g. codex) emit Responses SSE → translate into client format
const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES;
@@ -30,11 +31,11 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
if (needsCodexTranslation) {
const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI;
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
}
if (needsTranslation(targetFormat, sourceFormat)) {
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames);
return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials);
}
return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey);
@@ -43,7 +44,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent,
/**
* Handle streaming response — pipe provider SSE through transform stream to client.
*/
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) {
export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) {
if (onRequestSuccess) {
Promise.resolve()
.then(onRequestSuccess)
@@ -79,11 +80,16 @@ export async function handleStreamingResponse({ providerResponse, provider, mode
};
}
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey });
const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials });
// Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event
// Terminal bytes when the stream aborts after HTTP 200 was already sent, so the
// client sees a real error instead of a silently truncated stream.
// Responses passthrough keeps its own response.failed shape; every other client
// format gets the OpenAI error frame + [DONE], or `event: error` for Claude.
const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES;
const onAbortTerminal = isResponsesPassthrough ? buildAbortedResponsesTerminalBytes : null;
const onAbortTerminal = isResponsesPassthrough
? buildAbortedResponsesTerminalBytes
: (message) => buildStreamErrorBytes(HTTP_STATUS.GATEWAY_TIMEOUT, message, sourceFormat);
const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS;
const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs);
+26 -6
View File
@@ -2,13 +2,21 @@
import { randomUUID } from "node:crypto";
import { nowSec } from "./_base.js";
import { PROVIDERS } from "../../config/providers.js";
import { CODEX_CLI_VERSION } from "../../config/appConstants.js";
const CODEX_RESPONSES_URL = PROVIDERS["codex"].baseUrl;
const CODEX_USER_AGENT = "codex_cli_rs/0.136.0";
const CODEX_VERSION = "0.136.0";
const CODEX_USER_AGENT = `codex_cli_rs/${CODEX_CLI_VERSION}`;
const CODEX_ORIGINATOR = "codex_cli_rs";
const CODEX_MODEL_SUFFIX = "-image";
const CODEX_REF_DETAIL = "high";
const CODEX_IMAGES_MAIN_MODEL = "gpt-5.5";
const CODEX_TOOL_IMAGE_MODELS = new Set([
"gpt-image-1.5",
"gpt-image-2",
"gpt-image-2.5",
"gpt-image-2.5-flare",
"gpt-image-2.5-sunburst",
]);
function decodeAccountId(idToken) {
try {
@@ -27,6 +35,13 @@ function stripImageSuffix(model) {
return model.endsWith(CODEX_MODEL_SUFFIX) ? model.slice(0, -CODEX_MODEL_SUFFIX.length) : model;
}
function resolveCodexImageModels(model) {
if (CODEX_TOOL_IMAGE_MODELS.has(model)) {
return { responsesModel: CODEX_IMAGES_MAIN_MODEL, toolModel: model };
}
return { responsesModel: stripImageSuffix(model), toolModel: null };
}
function toDataUrl(input) {
if (!input || typeof input !== "string") return null;
if (/^data:image\//i.test(input) || /^https?:\/\//i.test(input)) return input;
@@ -157,7 +172,7 @@ export default {
"originator": CODEX_ORIGINATOR,
"session_id": randomUUID(),
"user-agent": CODEX_USER_AGENT,
"version": CODEX_VERSION,
"version": CODEX_CLI_VERSION,
"x-client-request-id": randomUUID(),
};
},
@@ -167,21 +182,26 @@ export default {
const single = toDataUrl(body.image);
if (single) refs.push(single);
const detail = body.image_detail || CODEX_REF_DETAIL;
const { responsesModel, toolModel } = resolveCodexImageModels(model);
const imgTool = { type: "image_generation", output_format: (body.output_format || "png").toLowerCase() };
if (toolModel) {
imgTool.action = refs.length > 0 ? "edit" : "generate";
imgTool.model = toolModel;
}
if (body.size && body.size !== "") imgTool.size = body.size;
if (body.quality && body.quality !== "") imgTool.quality = body.quality;
if (body.background && body.background !== "") imgTool.background = body.background;
return {
model: stripImageSuffix(model),
model: responsesModel,
instructions: "",
input: [{ type: "message", role: "user", content: buildContent(body.prompt, refs, detail) }],
tools: [imgTool],
tool_choice: "auto",
tool_choice: toolModel ? { type: "image_generation" } : "auto",
parallel_tool_calls: false,
prompt_cache_key: randomUUID(),
stream: true,
store: false,
reasoning: null,
reasoning: toolModel ? { effort: "medium", summary: "auto" } : null,
};
},
// Custom: codex parses SSE → either pipe to client or collect b64
+55 -12
View File
@@ -2,6 +2,7 @@ import { createErrorResult } from "../utils/error.js";
import { HTTP_STATUS } from "../config/runtimeConfig.js";
import { refreshTokenByProvider } from "../services/tokenRefresh.js";
import { PROVIDER_MEDIA } from "../providers/index.js";
import { getVideoAdapter } from "./videoProviders/index.js";
// Upstream fetch deadline for video job submission/polling (the job itself is
// async upstream — this only bounds the HTTP round-trip, not video rendering).
@@ -94,21 +95,49 @@ export async function handleVideoProxyCore({
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Unknown video action: ${action}`);
}
const method = requestId ? "GET" : "POST";
const url = buildUpstreamUrl(config, action, requestId);
const adapter = getVideoAdapter(provider);
const fetchSignal = combineSignals(signal, timeoutMs);
const doFetch = (token) =>
fetch(url, {
// Default (xAI shape) request plan; adapters override URL/method/headers/body.
const defaultPlan = () => {
const method = requestId ? "GET" : "POST";
return {
method,
headers: buildHeaders({ token, contentType: method === "POST" ? contentType : null, idempotencyKey: method === "POST" ? idempotencyKey : null }),
url: buildUpstreamUrl(config, action, requestId),
headers: buildHeaders({
token: credentials?.accessToken || credentials?.apiKey,
contentType: method === "POST" ? contentType : null,
idempotencyKey: method === "POST" ? idempotencyKey : null,
}),
body: method === "POST" ? rawBody : undefined,
signal: fetchSignal,
});
};
};
// Rebuilt per attempt so the auth retry below picks up the refreshed token.
const doFetch = async () => {
const plan = adapter
? await adapter.buildRequest({
config, action, requestId, rawBody, contentType, idempotencyKey, credentials, log,
token: credentials?.accessToken || credentials?.apiKey,
})
: defaultPlan();
if (plan.error) return { planError: plan.error };
return {
response: await fetch(plan.url, {
method: plan.method,
headers: plan.headers,
body: plan.body,
signal: fetchSignal,
}),
};
};
const method = requestId ? "GET" : "POST";
let upstream;
try {
upstream = await doFetch(credentials?.accessToken || credentials?.apiKey);
const first = await doFetch();
if (first.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${first.planError}`);
upstream = first.response;
} catch (error) {
if (error?.name === "AbortError" || error?.name === "TimeoutError") {
return createErrorResult(HTTP_STATUS.REQUEST_TIMEOUT, `[${provider}] video ${method} aborted: ${error.message}`);
@@ -136,7 +165,9 @@ export async function handleVideoProxyCore({
await upstream.body?.cancel?.();
} catch { /* noop */ }
try {
upstream = await doFetch(credentials.accessToken || credentials.apiKey);
const retry = await doFetch();
if (retry.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${retry.planError}`);
upstream = retry.response;
} catch (error) {
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video retry after refresh failed: ${error.message}`, credentials));
}
@@ -152,13 +183,25 @@ export async function handleVideoProxyCore({
return createErrorResult(upstream.status, `[${provider}] ${message.slice(0, 2000)}`);
}
// Success: pass the upstream JSON through untouched (request_id / status / video.url).
// Success: pass the upstream JSON through untouched (request_id / status / video.url),
// unless the adapter maps a provider-native shape onto it (Vertex operations).
let outBody = bodyText;
let outType = upstream.headers.get("content-type") || "application/json";
if (adapter?.transformResponse) {
try {
outBody = JSON.stringify(adapter.transformResponse(JSON.parse(bodyText)));
outType = "application/json";
} catch {
// Non-JSON or unexpected shape — fall back to the raw upstream body.
}
}
return {
success: true,
response: new Response(bodyText, {
response: new Response(outBody, {
status: upstream.status,
headers: {
"Content-Type": upstream.headers.get("content-type") || "application/json",
"Content-Type": outType,
"Access-Control-Allow-Origin": "*",
},
}),
+13
View File
@@ -0,0 +1,13 @@
// Video provider adapters.
//
// Default (no adapter) = xAI shape: raw body forwarded to {baseUrl}/{action},
// polled at {baseUrl}/{id}, upstream JSON passed through verbatim.
// A provider only needs an adapter when its wire format differs from that.
import openrouter from "./openrouter.js";
import vertex from "./vertex.js";
const ADAPTERS = { openrouter, vertex };
export function getVideoAdapter(provider) {
return ADAPTERS[provider] || null;
}
@@ -0,0 +1,39 @@
// OpenRouter video jobs — https://openrouter.ai/docs/api/api-reference/videos
//
// Same async shape as xAI (POST → { id, status }, GET → status/unsigned_urls),
// two differences only: creation POSTs to the collection root (no `/generations`
// suffix) and the account headers come from the registry entry.
// Response bodies are passed through verbatim.
// ponytail: generations only — OpenRouter has no edits/extensions endpoint today.
const SUPPORTED_ACTIONS = new Set(["generations"]);
function headers(config, token) {
return {
Accept: "application/json",
...(config.headers || {}),
...(token ? { Authorization: `Bearer ${token}` } : {}),
};
}
export default {
buildRequest({ config, action, requestId, rawBody, contentType, token }) {
const base = config.baseUrl.replace(/\/$/, "");
if (requestId) {
return { method: "GET", url: `${base}/${encodeURIComponent(requestId)}`, headers: headers(config, token) };
}
if (!SUPPORTED_ACTIONS.has(action)) {
return { error: `OpenRouter video supports 'generations' only (got '${action}')` };
}
if (contentType && !contentType.includes("application/json")) {
return { error: "OpenRouter video requires an application/json body" };
}
return {
method: "POST",
url: base,
headers: { ...headers(config, token), "Content-Type": "application/json" },
body: rawBody,
};
},
};
+159
View File
@@ -0,0 +1,159 @@
// Vertex AI (Veo) video jobs.
//
// Vertex does NOT speak the OpenAI-ish /v1/videos shape, so unlike OpenRouter
// this adapter translates both directions:
// create → POST {model}:predictLongRunning { instances[], parameters{} } → { name }
// poll → POST {model}:fetchPredictOperation { operationName } → { done, response }
// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation
//
// The operation name is a resource path (contains "/"), so it is base64url-encoded
// into the job id returned to the client — GET /v1/videos/{id} stays a flat path.
import { parseVertexSaJson, refreshVertexToken } from "../../services/tokenRefresh.js";
const DEFAULT_LOCATION = "us-central1";
const encodeJobId = (name) => Buffer.from(name, "utf8").toString("base64url");
// Operation name shape: projects/{p}/locations/{l}/publishers/{pub}/models/{m}/operations/{op}.
// Anchored and single-segment-per-field so a decoded path can never carry `..` or a
// host-changing prefix into the request URL.
const OPERATION_NAME_RE = /^projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/models\/[^/]+\/operations\/[^/]+$/;
function modelPathOf(operationName) {
return operationName.slice(0, operationName.indexOf("/operations/"));
}
function decodeJobId(id) {
const raw = String(id ?? "");
// Buffer.from(x, "base64url") silently drops invalid characters instead of
// throwing, so only ids that re-encode byte-for-byte are accepted.
if (!raw || raw.length > 1024 || !/^[A-Za-z0-9_-]+$/.test(raw)) return null;
const decoded = Buffer.from(raw, "base64url").toString("utf8");
if (Buffer.from(decoded, "utf8").toString("base64url") !== raw) return null;
return OPERATION_NAME_RE.test(decoded) ? decoded : null;
}
async function resolveAuth(credentials, log) {
const saJson = parseVertexSaJson(credentials?.apiKey);
const projectId =
saJson?.project_id ||
credentials?.projectId ||
credentials?.providerSpecificData?.projectId;
const location = credentials?.providerSpecificData?.location || DEFAULT_LOCATION;
if (!projectId) {
return { error: "Vertex video requires a project_id — use Service Account JSON or set providerSpecificData.projectId" };
}
let token = credentials?.accessToken;
if (saJson) {
const minted = await refreshVertexToken(saJson, log);
if (!minted?.accessToken) return { error: "Vertex video: failed to mint access token from service account JSON" };
token = minted.accessToken;
}
if (!token) return { error: "Vertex video requires Service Account JSON or an OAuth access token (raw API keys are not supported)" };
return { token, projectId, location };
}
/** OpenAI-ish video body → Vertex predictLongRunning body. */
function toVertexBody(body) {
const instance = { prompt: body.prompt };
// Image-to-video: accept the Vertex-native shape or a bare data URL / base64 string.
const image = body.image ?? body.image_url;
if (image && typeof image === "object") {
instance.image = image;
} else if (typeof image === "string") {
const match = image.match(/^data:([^;]+);base64,(.*)$/s);
instance.image = match
? { bytesBase64Encoded: match[2], mimeType: match[1] }
: { gcsUri: image };
}
if (body.video && typeof body.video === "object") instance.video = body.video;
const parameters = {};
if (body.n != null) parameters.sampleCount = Number(body.n);
if (body.duration != null) parameters.durationSeconds = Number(body.duration);
if (body.aspect_ratio) parameters.aspectRatio = body.aspect_ratio;
if (body.resolution) parameters.resolution = body.resolution;
if (body.seed != null) parameters.seed = body.seed;
if (body.negative_prompt) parameters.negativePrompt = body.negative_prompt;
// Without storageUri Vertex returns inline base64 bytes; a GCS bucket keeps
// the poll response small and is what production callers want.
if (body.storage_uri) parameters.storageUri = body.storage_uri;
if (body.generate_audio != null) parameters.generateAudio = !!body.generate_audio;
return { instances: [instance], ...(Object.keys(parameters).length ? { parameters } : {}) };
}
/** Vertex operation → the async-job shape 9Router clients already poll for. */
function fromVertexOperation(json) {
if (!json?.name) return json;
const id = encodeJobId(json.name);
if (json.error) {
return { id, request_id: id, status: "failed", error: json.error };
}
if (!json.done) {
return { id, request_id: id, status: "pending" };
}
const samples =
json.response?.videos ||
json.response?.generateVideoResponse?.generatedSamples ||
[];
const videos = samples.map((s) => ({
url: s.gcsUri || s.video?.uri || s.uri || null,
b64_json: s.bytesBase64Encoded || s.video?.bytesBase64Encoded || null,
mime_type: s.mimeType || s.video?.mimeType || "video/mp4",
}));
return { id, request_id: id, status: "completed", video: videos[0] || null, videos };
}
export default {
async buildRequest({ config, action, requestId, rawBody, contentType, credentials, log }) {
if (contentType && !contentType.includes("application/json")) {
return { error: "Vertex video requires an application/json body" };
}
const auth = await resolveAuth(credentials, log);
if (auth.error) return { error: auth.error };
const { token, projectId, location } = auth;
const base = (config.baseUrl || "https://aiplatform.googleapis.com").replace(/\/$/, "");
const headers = { Accept: "application/json", "Content-Type": "application/json", Authorization: `Bearer ${token}` };
if (requestId) {
const operationName = decodeJobId(requestId);
if (!operationName) return { error: "Invalid Vertex video job id" };
return {
method: "POST",
url: `${base}/v1/${modelPathOf(operationName)}:fetchPredictOperation`,
headers,
body: JSON.stringify({ operationName }),
};
}
if (action !== "generations") {
// ponytail: Veo extend/edit go through generations with `video`/`image` in the body.
return { error: `Vertex video supports 'generations' only (got '${action}')` };
}
let body;
try {
body = JSON.parse(typeof rawBody === "string" ? rawBody : rawBody.toString("utf8"));
} catch {
return { error: "Invalid JSON body" };
}
if (!body.model) return { error: "Vertex video requires a model (e.g. vertex/veo-3.1-generate-preview)" };
// Plain model id only — a path segment carrying "/" or ".." would rewrite the URL.
if (!/^[A-Za-z0-9._-]+$/.test(body.model)) return { error: "Invalid Vertex video model id" };
if (!body.prompt && !body.image && !body.image_url) return { error: "Vertex video requires a prompt or an image" };
return {
method: "POST",
url: `${base}/v1/projects/${projectId}/locations/${location}/publishers/google/models/${body.model}:predictLongRunning`,
headers,
body: JSON.stringify(toVertexBody(body)),
};
},
transformResponse: fromVertexOperation,
};
+169 -25
View File
@@ -116,6 +116,16 @@ export const MODEL_CAPABILITIES = {
// DeepSeek's first V4 model with image input; text limits match V4-Flash.
"deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
// DeepSeek V4.1-Flash is natively multimodal — models.dev lists
// opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream
// the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the
// same image capability as the exp id above. "deepseek-flash" is the GA id on the
// DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose
// 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry
// short-circuits the pattern table, so a vision-only delta would drop them.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
"deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 },
// Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases
"vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
"coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 },
@@ -131,6 +141,8 @@ export const MODEL_CAPABILITIES = {
// via OpenAI Responses input_image; reasoning supports up to xhigh.
"muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
"muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 },
// OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output
"union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 },
};
const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 };
@@ -154,6 +166,7 @@ export const PROVIDER_CAPABILITIES = {
"deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 },
},
"codex": {
"gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 },
"gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS,
"gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS,
@@ -178,39 +191,95 @@ export const PROVIDER_CAPABILITIES = {
// CodeBuddy.cn — authoritative per-model metadata from the gateway's model
// config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision=
// supportsImages). Every model reasons via OpenAI-style reasoning_effort
// (see registry thinkingFormat). `onlyReasoning` models can't turn thinking
// off → thinkingCanDisable:false (clamped to minimal instead of disabled).
// (see registry thinkingFormat). For thinkingCanDisable use the server's
// reasoning.canDisableThinking flag — see the note in the codebuddy-cn block
// below; it is NOT the inverse of onlyReasoning.
"codebuddy-cn": {
"glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 },
// maxOutput 64000 per both the plugin-baked fallback and the live server
// table (the old 38000 had no source and truncated output).
"glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 },
"glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 },
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 },
"minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 },
"kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 },
"hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
// hy3/hy3-x: 256K official (192K conservative, matches hy3-preview); hy4-preview: 1M official.
// glm-5.3: 1M (GLM-5.x gen); glm-5.3-flash window unverified (200K conservative).
// Per-model values mirror the server's product-config payload (the plugin
// fetches it from copilot.tencent.com; the `models[]` entries carry
// maxInputTokens/maxOutputTokens/supportsImages). contextWindow =
// maxInputTokens, maxOutput = maxOutputTokens. Where the server and the
// plugin-baked fallback disagree, the server table wins.
// ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking —
// it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is
// on by default"; canDisableThinking means "it CAN be turned off". glm-5.3
// and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so
// their thinking is switchable; the hy* models are forced always-on.
"hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"hy3-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 },
"hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
"hy4-preview-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 },
"glm-5.3": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.3-flash": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 },
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 },
"deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 },
"glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 },
"glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 },
"kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 },
"deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 },
// deepseek-v4.1-flash replaces v4-flash (dropped from the server list;
// the old endpoint still answers 200 but the published list is the
// contract). maxOutput 128000 per the server's product-config payload.
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors
// the codebuddy-cn entry (the openai-style reasoning_effort format matters:
// the generic *deepseek-v4* pattern would otherwise pick the vendor-native
// "deepseek" thinking shape, which the CodeBuddy gateway does not accept).
"codebuddy-intl": {
"deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 },
},
// Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the
// registry `name` is display-only and capability lookup matches on the raw
// id, so every qoder model would fall through to DEFAULT_CAPABILITIES
// (200K) without this map. contextWindow follows the real model family's
// spec: the /algo/api/v2/model/list max_input_tokens under-reports some
// windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more).
// max_output_tokens arrives as 0 for every model, so outputs are
// best-guess from the real model family. Vision tags below follow the
// upstream is_vl flag. The executor uploads inlined images to
// /api/v2/image/upload and leaves image_urls/chat_context.imageUrls null
// (same as qodercli). reasoning:true on all of them — every model can
// reason; the upstream is_reasoning flag only drives model_config selection.
// thinkingFormat keeps the true-model family for documentation/UI, but
// thinkingCanDisable:false everywhere: the executor only forwards
// messages/tools/max_tokens, and thinking is fixed upstream via
// modelConfig.is_reasoning — client thinking intent is dropped, so "none"
// must never be offered as an option.
"qoder": {
"ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5
"performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5
"dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro
"dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash
"gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3
"gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash
"kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3
"kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code
"mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3
"qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max
"qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus
"qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash
"qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max
},
// Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output).
"poolside": {
"laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 },
"laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 },
},
// Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge
// the library page publishes for this model (text+image in, 1M context).
// ponytail: thinkingFormat stays "deepseek" to preserve today's body shape;
// Ollama's native toggle is the top-level `think` field (bool or
// low/medium/high/max), which no format in thinkingUnified.js emits yet —
// openai-to-ollama.js drops it. Wire a "think" format when thinking on
// Ollama Cloud is actually needed.
"ollama": {
"deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 },
},
};
/**
@@ -247,6 +316,9 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } },
{ pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } },
// ── OpenAI GPT-6.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } },
// ── OpenAI GPT-5.x (vision + thinking + web search) ──────────────
{ pattern: "*gpt-5*image*", caps: { imageOutput: true } },
{ pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } },
@@ -312,7 +384,11 @@ export const PATTERN_CAPABILITIES = [
{ pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } },
// ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } },
// v4.1+ has real image input (probed live on Alibaba MaaS: correct color
// read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore
// them (answered "Unknown"), so vision stays scoped to v4.* dotted releases.
{ pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } },
{ pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } },
{ pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
{ pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } },
{ pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } },
@@ -377,14 +453,27 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"];
// Catalog lookups, installed by the server at startup. Left as no-ops in the
// browser bundle, where there is no file to read.
//
// The server bundles this module into every route chunk that needs it, and each
// copy carries its own module state, so an install landing in the copy the
// startup hook imported stays invisible to the copy resolving requests. The slot
// lives on globalThis instead; the local binding is the fast path.
let catalogSource = null;
/**
* Install the synced catalog reader (server only).
* @param {{ getModalities: Function, getLimits: Function } | null} source
* @param {{ getModalities: (provider: string, model: string) => object|null,
* getLimits: (provider: string, model: string) => object|null } | null} source
*/
export function setCatalogSource(source) {
catalogSource = source;
if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source;
}
function getCatalogSource() {
if (catalogSource) return catalogSource;
if (typeof globalThis === "undefined") return null;
return (catalogSource = globalThis.__9rCatalogSource || null);
}
// Apply the synced catalog + name heuristic on top of a table-resolved result.
@@ -393,15 +482,16 @@ export function setCatalogSource(source) {
function refine(base, provider, model) {
const result = { ...DEFAULT_CAPABILITIES, ...base };
if (catalogSource) {
const modalities = catalogSource.getModalities(model);
const source = getCatalogSource();
if (source) {
const modalities = source.getModalities(provider, model);
if (modalities) {
for (const key of MODALITY_KEYS) {
if (modalities[key] === true) result[key] = true;
}
}
const limits = catalogSource.getLimits(provider, model);
const limits = source.getLimits(provider, model);
if (limits) {
if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow;
if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput;
@@ -413,12 +503,66 @@ function refine(base, provider, model) {
return result;
}
// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models
// default to vision; only this denylist stays text-only.
const COMMANDCODE_TEXT_ONLY = new Set([
"deepseek/deepseek-v4-pro",
"deepseek/deepseek-v4-flash",
"deepseek/deepseek-v4-flash-fast",
"zai-org/glm-5.3",
"zai-org/glm-5.2",
"zai-org/glm-5.2-fast",
"zai-org/glm-5.1",
"zai-org/glm-5",
"minimaxai/minimax-m2.7",
"minimax/minimax-m2.7-free",
"minimaxai/minimax-m2.5",
"xiaomi/mimo-v2.5-pro",
"qwen/qwen3.6-max-preview",
"qwen/qwen3.7-max",
"meituan/longcat-2.0:free",
"stepfun/step-3.5-flash",
"tencent/hy4-preview",
"tencent/hy3",
"tencent/hy3-paid",
"nvidia/nemotron-3-ultra-550b-a55b",
"poolside/laguna-s-2.1-free",
"inclusionai/ling-3.0-flash-free",
"inclusionai/ling-3.0-flash-sante:free",
]);
function isCommandCodeTextOnly(model) {
const key = String(model || "").toLowerCase();
if (COMMANDCODE_TEXT_ONLY.has(key)) return true;
for (const id of COMMANDCODE_TEXT_ONLY) {
const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
if (key === base || key.endsWith("/" + base)) return true;
}
return false;
}
export function getCapabilitiesForModel(provider, model) {
if (!model) return { ...DEFAULT_CAPABILITIES };
// Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7".
const baseModel = model.includes("/") ? model.split("/").pop() : model;
// CommandCode wire is /alpha/generate for every model. Family patterns
// (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here.
if (provider === "commandcode" || provider === "cmc") {
const providerCaps = PROVIDER_CAPABILITIES.commandcode;
if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] };
if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] };
return {
...DEFAULT_CAPABILITIES,
reasoning: true,
thinkingFormat: "commandcode",
thinkingEffortSupported: true,
vision: !isCommandCodeTextOnly(model),
contextWindow: 1000000,
maxOutput: 384000,
};
}
// 1. Provider-specific override
if (provider) {
const providerCaps = PROVIDER_CAPABILITIES[provider];
+16 -6
View File
@@ -13,6 +13,11 @@ export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json");
// Trimmed upstream catalog, read by the add-models skill (not by the router).
export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json");
// Schema of the file this module reads. The writer stamps it; a file carrying an
// older value predates provider-scoped modality keys, and its flat keys are not
// looked up here, so the sync rebuilds it instead of asking upstream for a 304.
export const CATALOG_VERSION = 2;
const EMPTY = { models: {}, providers: {} };
let cache = EMPTY;
let cachedMtime = -1;
@@ -45,14 +50,19 @@ function load() {
return cache;
}
// Modality is a property of the model itself — any gateway serving it inherits
// the same image/video/pdf support, so this is keyed by model id alone.
export function getCatalogModalities(model) {
return load().models[baseId(model)] || null;
// Modalities are recorded per gateway upstream, and gateways disagree about the
// same weights — some do not proxy images at all — so the key is provider +
// model, in the local provider id space, exactly like the limits below. Keying
// by model id alone made short ids collide across vendors: "auto", "free" and
// "efficient" are router modes in one catalog and model names in another, and a
// request to the router mode inherited a stranger's vision.
export function getCatalogModalities(provider, model) {
if (!provider) return null;
return load().models[`${provider}:${baseId(model)}`] || null;
}
// Context and output limits are a property of the gateway, not the model: each
// one truncates differently, so these stay keyed by provider + model.
// Context and output limits are a property of the gateway too: each one
// truncates differently, so these stay keyed by provider + model.
export function getCatalogLimits(provider, model) {
const byProvider = provider && load().providers[provider];
if (!byProvider) return null;
+3
View File
@@ -53,6 +53,7 @@ export const MODEL_PRICING = {
"gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 },
"gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 },
"gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 },
"o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 },
"o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 },
@@ -110,6 +111,8 @@ export const MODEL_PRICING = {
"deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 },
"deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 },
// === GLM ===
@@ -37,6 +37,7 @@ export default {
},
usage: {
quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`,
quotaSummaryApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:retrieveUserQuotaSummary`,
loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist",
tokenUrl: "https://oauth2.googleapis.com/token",
},
+6 -3
View File
@@ -20,6 +20,8 @@ export default {
authModes: [
"apikey",
],
passthroughModels: true,
modelsFetcher: { url: "https://api.airforce/v1/models", type: "airforce-free" },
transport: {
baseUrl: "https://api.airforce/v1/chat/completions",
validateUrl: "https://api.airforce/v1/models",
@@ -27,10 +29,11 @@ export default {
"HTTP-Referer": "https://endpoint-proxy.local",
"X-Title": "Endpoint Proxy",
},
forceStream: true,
},
models: [
{ id: "anthropic/claude-3.7-sonnet", name: "Claude 3.7 Sonnet (Free)", contextLength: 200000 },
{ id: "moonshot/kimi-k2.6", name: "Kimi K2.6 (Free)", contextLength: 262144 },
{ id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash (Free)", contextLength: 1048576 },
{ id: "gpt-oss-120b", name: "GPT-OSS 120B (Free)", contextLength: 131072 },
{ id: "gpt-oss-20b", name: "GPT-OSS 20B (Free)", contextLength: 131072 },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code (Free)", contextLength: 262144 },
],
};
+2
View File
@@ -23,6 +23,8 @@ export default {
"HTTP-Referer": "https://cline.bot",
"X-Title": "Cline",
},
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
quirks: { clineEnvelope: true },
tokenUrl: "https://api.cline.bot/api/v1/auth/token",
refreshUrl: "https://api.cline.bot/api/v1/auth/refresh",
auth: {
+6 -1
View File
@@ -14,7 +14,10 @@ export default {
},
},
category: "oauth",
authModes: ["oauth", "apikey"],
// ClinePass authenticates with a plain API key from app.cline.bot/settings/api-keys
// (category "apikey"). The OAuth extension flow used by Cline does not issue
// tokens that the ClinePass API consumer endpoint accepts (HTTP 401) — see #2333.
authModes: ["apikey", "oauth"],
hasOAuth: true,
transport: {
baseUrl: "https://api.cline.bot/api/v1/chat/completions",
@@ -22,6 +25,8 @@ export default {
"HTTP-Referer": "https://cline.bot",
"X-Title": "Cline",
},
// Non-stream chat completions come back wrapped in {"success":true,"data":{...}}
quirks: { clineEnvelope: true },
auth: {
combined: true,
header: "Authorization",
+12 -11
View File
@@ -47,27 +47,28 @@ export default {
models: [
{ id: "glm-5.2", name: "GLM-5.2" },
{ id: "glm-5.1", name: "GLM-5.1" },
{ id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" },
{ id: "glm-5v-turbo", name: "GLM-5v-Turbo" },
{ id: "minimax-m3", name: "MiniMax-M3" },
{ id: "minimax-m2.7", name: "MiniMax-M2.7" },
{ id: "kimi-k2.7", name: "Kimi-K2.7-Code" },
{ id: "kimi-k2.6", name: "Kimi-K2.6" },
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
// "-x" suffix = paid tier of the same model (free id rides the promo quota:
// hy3 free until 2026-08-31, hy4-preview until 2026-09-10). Server model table
// seen in client logs 2026-08-30; glm-5.0 / glm-4.7 removed (API 11102 dead).
{ id: "hy3-preview", name: "Hy3 Preview" },
// Catalog mirrors the server's product-config payload (the plugin fetches
// it from copilot.tencent.com). Models the server no longer publishes are
// removed even when the chat endpoint still answers them — the published
// list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x
// (endpoint returns 11102 "model service info not found"), plus
// glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview /
// deepseek-v3-2-volc (absent from the server list, though still answering
// 200) and hy3-x (paid tier, not used here). deepseek-v4-flash removed
// 2026-09: replaced server-side by deepseek-v4.1-flash (same low/high/
// xhigh efforts; endpoint still answers 200 but the list is the contract).
// "-x" suffix = paid tier of the same model (free id rides the promo quota).
{ id: "hy3", name: "Hy3" },
{ id: "hy3-x", name: "Hy3 (Paid)" },
{ id: "hy4-preview", name: "Hy4-Preview" },
{ id: "hy4-preview-x", name: "Hy4-Preview (Paid)" },
{ id: "glm-5.3", name: "GLM-5.3" },
{ id: "glm-5.3-flash", name: "GLM-5.3-Flash" },
{ id: "kimi-k3-1", name: "Kimi-K3" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
],
oauth: {
baseUrl: "https://copilot.tencent.com",
@@ -58,7 +58,9 @@ export default {
{ id: "kimi-k2.5", name: "Kimi-K2.5" },
{ id: "hy3-preview", name: "Hy3 Preview" },
{ id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" },
{ id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" },
// deepseek-v4-flash replaced server-side by deepseek-v4.1-flash (same
// catalog as CN; the old endpoint still answers 200 but the list is the contract).
{ id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" },
{ id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" },
],
oauth: {
+18 -1
View File
@@ -1,5 +1,9 @@
import { withCodexReviewModels } from "../models/helpers.js";
// Codex CLI version seen by OpenAI's backend — single source for the Version /
// User-Agent identity headers. Bump when the installed codex CLI is upgraded.
const CODEX_CLI_VERSION = "0.154.0";
export default {
id: "codex",
priority: 30,
@@ -34,9 +38,10 @@ export default {
baseUrl: "https://chatgpt.com/backend-api/codex/responses",
format: "openai-responses",
forceStream: true,
cliVersion: CODEX_CLI_VERSION,
headers: {
originator: "codex_cli_rs",
"User-Agent": "codex_cli_rs/0.136.0",
"User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`,
},
usage: {
url: "https://chatgpt.com/backend-api/wham/usage",
@@ -45,6 +50,7 @@ export default {
},
},
models: [
{ id: "gpt-6-astra", name: "GPT 6.0 Astra" },
{ id: "gpt-5.6-sol", name: "GPT 5.6 Sol" },
{ id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" },
{ id: "gpt-5.6-terra", name: "GPT 5.6 Terra" },
@@ -59,6 +65,17 @@ export default {
{ id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" },
{ id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" },
{ id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" },
// Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived
// from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398).
{ id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" },
{ id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-2", name: "GPT Image 2", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-image-1.5", name: "GPT Image 1.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
{ id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" },
@@ -40,4 +40,8 @@ export default {
{ id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" },
{ id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" },
],
features: {
usage: true,
usageApikey: true,
},
};
+16
View File
@@ -25,6 +25,21 @@ export default {
reasoningInject: {
scope: "all",
},
quirks: {
// DeepSeek's Anthropic-compatible endpoint
// (https://api.deepseek.com/anthropic/v1/messages) accepts ONLY the
// built-in web_search_* tools and rejects client-defined `custom` tools
// (MCP / Read / Bash / etc.) with HTTP 400
// "tools[0]: unknown variant `custom`, expected
// `web_search_20250305` or `web_search_20260209`".
//
// Declaring this whitelist makes prepareClaudeRequest() forward only
// web_search_* tools and strip everything else before sending, so MCP /
// function tools are dropped instead of failing the whole request.
// DeepSeek's OpenAI-compatible transport is unaffected (targetFormat
// there is "openai", not "claude", so prepareClaudeRequest is not run).
claudeSupportedToolTypes: ["web_search_20250305", "web_search_20260209"],
},
},
// Multi-endpoint: pick the transport matching client sourceFormat to skip translation.
transports: [
@@ -44,6 +59,7 @@ export default {
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" },
{ id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" },
{ id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash" },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" },
{ id: "deepseek-chat", name: "DeepSeek V3.2 Chat" },
+1
View File
@@ -25,6 +25,7 @@ export default {
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM-4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
+1
View File
@@ -49,6 +49,7 @@ export default {
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" },
{ id: "glm-5.2", name: "GLM 5.2" },
{ id: "glm-5.1", name: "GLM 5.1" },
{ id: "glm-5-turbo", name: "GLM 5 Turbo" },
{ id: "glm-5", name: "GLM 5" },
{ id: "glm-4.7", name: "GLM 4.7" },
{ id: "glm-4.6v", name: "GLM 4.6V (Vision)" },
@@ -22,6 +22,7 @@ export default {
headers: { ...CLAUDE_API_HEADERS },
quirks: {
dropOutputConfig: true,
requireClaudeToolType: true,
},
reasoningInject: {
scope: "all",
+1
View File
@@ -22,6 +22,7 @@ export default {
headers: { ...CLAUDE_API_HEADERS },
quirks: {
dropOutputConfig: true,
requireClaudeToolType: true,
},
reasoningInject: {
scope: "all",
+1
View File
@@ -30,6 +30,7 @@ export default {
{ id: "glm-4.7-flash", name: "GLM 4.7 Flash" },
{ id: "qwen3.5", name: "Qwen3.5" },
{ id: "minimax-m3", name: "MiniMax M3" },
{ id: "deepseek-v4.1-flash:cloud", name: "DeepSeek V4.1 Flash" },
],
serviceKinds: ["llm", "webFetch"],
fetchConfig: {
+3
View File
@@ -57,6 +57,9 @@ export default {
{ id: "whisper-1", name: "Whisper 1", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-4o-transcribe", name: "GPT-4o Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-4o-mini-transcribe", name: "GPT-4o Mini Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" },
{ id: "gpt-image-2.5", name: "GPT Image 2.5", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "gpt-image-1", name: "GPT Image 1", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "dall-e-3", name: "DALL-E 3", params: ["size","quality","style","response_format"], kind: "image" },
{ id: "dall-e-2", name: "DALL-E 2", params: ["n","size","response_format"], kind: "image" },
+23 -1
View File
@@ -13,7 +13,7 @@ export default {
textIcon: "OC",
website: "https://opencode.ai/auth",
notice: {
text: "OpenCode Go subscription: $5/mo (then 0/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
text: "OpenCode Go subscription: $5/mo (then 10/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.",
apiKeyUrl: "https://opencode.ai/auth",
},
},
@@ -21,6 +21,9 @@ export default {
transport: {
baseUrl: "https://opencode.ai/zen/go/v1/chat/completions",
headers: {},
usage: {
url: "https://opencode.ai/zen/go/v1/usage",
},
},
// Multi-endpoint: pick the transport matching the client sourceFormat to skip
// translation. Guarded per-model by `supportedFormats` (see chatCore) because
@@ -30,22 +33,41 @@ export default {
{ format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } },
{ format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } },
],
// supportedFormats follow the endpoint table in https://opencode.ai/docs/go/
models: [
{ id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] },
{ id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] },
{ id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] },
{ id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] },
{ id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] },
{ id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] },
{ id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] },
{ id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] },
{ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] },
{ id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] },
{ id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] },
{ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] },
{ id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] },
{ id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] },
{ id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] },
{ id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] },
{ id: "hy3", name: "Hy3", supportedFormats: ["openai"] },
// Served by /zen/go/v1/responses only — the responses-only entry forces chatCore
// past the sourceFormat-matched transports into translation (see chatCore guard).
{ id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
{ id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] },
],
features: {
usage: true,
usageApikey: true,
},
};
+6 -2
View File
@@ -17,13 +17,17 @@ export default {
headers: {
"x-opencode-client": "desktop",
},
forceStream: true,
noAuth: true,
quirks: {
forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"],
},
},
models: [
// Muse Spark models are served by /zen/v1/responses; the rest stay on
// /chat/completions, so the format is declared per-model, not per-provider.
// Endpoint formats differ per model, so declare non-chat models explicitly.
{ id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" },
{ id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" },
{ id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" },
],
modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" },
passthroughModels: true,
+10 -1
View File
@@ -40,8 +40,11 @@ export default {
{ id: "openai/gpt-image-1", name: "GPT Image 1 (via OpenRouter)", params: ["n","size","quality","response_format"], kind: "image" },
{ id: "google/imagen-3.0-generate-002", name: "Imagen 3 (via OpenRouter)", params: ["n","size"], kind: "image" },
{ id: "black-forest-labs/FLUX.1-schnell", name: "FLUX.1 Schnell (via OpenRouter)", params: ["n","size"], kind: "image" },
{ id: "google/veo-3.1", name: "Veo 3.1 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "openai/sora-2-pro", name: "Sora 2 Pro (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
{ id: "bytedance/seedance-2.0", name: "Seedance 2.0 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" },
],
serviceKinds: ["llm","embedding","tts","imageToText"],
serviceKinds: ["llm","embedding","tts","imageToText","video"],
ttsConfig: {
baseUrl: "https://openrouter.ai/api/v1/chat/completions",
defaultModel: "openai/gpt-4o-mini-tts",
@@ -57,6 +60,12 @@ export default {
baseUrl: "https://openrouter.ai/api/v1/images/generations",
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
},
// Async video jobs (POST /videos → { id, status }, GET /videos/{id} polls).
// Docs: https://openrouter.ai/docs/api/api-reference/videos
videoConfig: {
baseUrl: "https://openrouter.ai/api/v1/videos",
headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"},
},
modelsFetcher: { url: "https://openrouter.ai/api/v1/models", type: "openrouter-free" },
passthroughModels: true,
};

Some files were not shown because too many files have changed in this diff Show More