diff --git a/.gitignore b/.gitignore index 57f710a5..c9d3e04a 100644 --- a/.gitignore +++ b/.gitignore @@ -89,3 +89,15 @@ graphify-out/* # Kiro local workspace state .kiro/ +9router-* + +# Local sensitive / temp files +.engine-token.txt +.tmp-prov.json +.tmp-providers.json +.oauth-session.json +.dev-server.log +.dev-server-err.log +.start-dev.ps1 +start-dev-silent.cjs +debug.log diff --git a/CHANGELOG.md b/CHANGELOG.md index da6b2c05..f16c8f53 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,71 @@ +# v0.5.81 (2026-09-18) + +## Features +- **Xiaomi MiMo**: merge MiMo Desktop support into `xiaomi-mimo` with dual auth (API key + Desktop/OAuth session), Preview models support, and encrypted-callback OAuth flow +- **Claude Code**: add 1M-context toggle (`[1m]` marker) and drive `CLAUDE_CODE_AUTO_COMPACT_WINDOW` directly from the dashboard +- **Models**: add DeepSeek-V4.1-Flash to DeepSeek provider, CodeBuddy-Intl, and Ollama (`deepseek-v4.1-flash:cloud`); enable `low`..`max` reasoning effort levels and vision capability for DeepSeek-V4.* +- **i18n**: integrate Persian (fa) translation + +## Fixes +- **OpenCode / OpenCode Go**: resolve 403 `FreeTierError` and 429 rate limits with canonical session format, valid User-Agent, and stable upstream session reuse; force stream and declare `forceStream` for free-tier SSE aggregation; cloak decoy tools, normalize Muse Free tool choice, and strip prior reasoning items on Responses models; route Union Alpha via Messages API +- **Kiro**: preserve underscores in tool names (`mcp__server__tool`) and restore client tool names in responses; use neutral placeholder for tool-result-only turns; forward tool-result images +- **Stream**: report aborts after HTTP 200 in-band (per-format error frames) instead of closing silently +- **Command Code**: preserve images and `reasoning_effort` on `/alpha/generate`; retry transient stream errors and avoid fake stop chunks; add Quota Tracker support +- **Zed**: harden OAuth lifecycle (preserve `systemId`, renew proxy timeout), support live model resolution, and lower display priority in OAuth list +- **Antigravity**: scope cached thought signatures to model family; strip Claude Code billing headers from system prompts; sanitize Hermes system identity +- **Codex**: route bare `codex-auto-review` requests to the Codex provider (#4135) +- **Auth**: do not cool down an account for request-scoped 4xx errors +- **Usage**: improve DeepSeek credit balance display as currency credit instead of 0/total quota bar +- **Model Catalog**: scope synced catalog to gateways and declare vision capabilities for DeepSeek V4.1-Flash IDs + +# v0.5.75 (2026-09-10) + +## Features +- **Video**: add OpenRouter and Vertex AI (Veo) video generation on `/v1/videos/*` via a provider adapter layer; poll requests resolve their provider from `x-connection-id` or `?provider=` +- **Antigravity**: add weekly quota tracking (Gemini weekly / Claude & GPT weekly) and free-tier handling from `retrieveUserQuotaSummary` (#3892) +- **Codex**: add GPT Image 2.5, Flare and Sunburst image models with multi-image support; add the same ids to the OpenAI catalog +- **Qoder**: surface usage to all clients and stop inlining large attachments — images upload through `/api/v2/image/upload` like qodercli, oversized file blocks become stubs, context tier auto-escalates +- **OpenCode Go**: add newly published models (glm-5.3, kimi-k3, deepseek-flash, longcat-2.0, hy4-preview, hy3 on chat/completions; qwen3.8-max, qwen3.8-flash on `/messages`; grok-4.6, gpt-5.6-luna on Responses) and list `deepseek-v4.1-flash` first in the catalog +- **CLI tools**: group the model selector by provider with full-text search and manual custom model ID entry +- **CodeBuddy-CN**: replace `deepseek-v4-flash` with `deepseek-v4.1-flash` + +## Fixes +- **Tools**: scope Claude tool type defaulting to gateways declaring `requireClaudeToolType` — the global default broke Anthropic-compatible endpoints that only accept the legacy typeless tool shape (#3905) +- **Claude**: cap re-anchored `cache_control` at the 4-marker budget so a spent budget no longer 400s and triggers a full combo failover; wrap bare single-object content turns before the mid-conversation-system fold +- **Cline / Airforce**: unwrap the `{"success":true,"data":…}` envelope on non-stream chat completions (#3644); add the live Cline/ClinePass model catalog and refresh Airforce free models +- **Cline**: stop `workos:`-prefixing ClinePass API keys (401 on every request, #2333) and add clinepass token refresh +- **Kiro**: never send a top-level `systemPrompt` (`400 REQUEST_BODY_INVALID`); route requests through current runtime surfaces (#3776) +- **Codex**: strip Unicode-property tool schema patterns the validator rejects (#3922); restore the `Version` header and single-source the CLI version +- **DeepSeek**: keep Anthropic-only tool types when forwarding to `/anthropic/v1/messages` +- **Qoder**: drop the Responses usage plumbing from shared translator/handler code, which changed token accounting for every provider, not just Qoder +- **Antigravity**: normalize contents and handle intermediate tool responses; protect the OAuth token-refresh path from Google anti-abuse rate limits (#3813) +- **Providers**: clear stale connection health state (`modelLock_*`, `backoffLevel`, `rateLimitedUntil`, `errorCode`) when a connection is re-validated (#3810, #3830); remove the duplicate `qwen` provider that shadowed `alims-intl` +- **Video / Vertex**: reject job ids and model ids that would escape the request URL path (SSRF) +- **Usage**: parse the Fable weekly limit from `limits[]` instead of fabricating a row (#3847) +- **Auth**: set a 24h `maxAge` on the dashboard session cookie + +# v0.5.69 (2026-09-05) + +## Features +- **Codex**: add GPT 6.0 Astra (`gpt-6-astra`) with vision, thinking and search capabilities +- **Usage**: add Claude Fable quota tracker support with weekly window normalization (`weekly fable (7d)`) +- **Dashboard**: group Antigravity Gemini and Claude quotas in Quota Tracker, prune stale hidden keys +- **OpenCode Go**: add `muse-spark-1.3-contributor` model and support parallel tool calls on Responses path (#3819) +- **Providers & Models**: align CodeBuddy-CN catalog/capabilities with server config; add GPT-5.6 Sol, Terra, Luna image aliases on Codex (#3806); refresh Qoder catalog with capability mapping and image pass-through +- **CLI tools**: replace Copilot MITM with VS Code extension setup guide +- **Gemini**: persist and replay `thoughtSignature` scoped by session namespace + +## Fixes +- **Claude**: normalize adaptive auto effort (`output_config.effort`) (#3792) +- **Antigravity**: prevent Google anti-abuse rate limits during multi-account refresh (#3813) +- **Anthropic-compatible**: forward Claude beta flags to nodes fronting Anthropic (#3797) +- **Dashboard**: dynamic mode label for local/remote detection (#3801) +- **Codex**: format reset credit API errors cleanly (#3778) +- **Security**: guard cowork MCP tools probe against SSRF (#3783) +- **OpenCode Go**: track OpenCode Go quota (#3791) and send stable session headers (#3800) +- **Logger**: suppress noisy background token refresh logs +- **CLI**: export packed `.tgz` directly into workspace root instead of parent directory + # v0.5.65 (2026-09-03) ## Features diff --git a/cli/package.json b/cli/package.json index d5776b4e..c75e6dc9 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "9router", - "version": "0.5.65", + "version": "0.5.81", "description": "9Router CLI - Start and manage 9Router server", "bin": { "9router": "./cli.js" @@ -16,7 +16,7 @@ "scripts": { "dev": "nodemon -I --watch cli.js --watch src --watch hooks --ext js,json cli.js", "build": "node scripts/build-cli.js", - "pack:cli": "npm run build && npm pack --pack-destination ../..", + "pack:cli": "npm run build && npm pack --pack-destination ..", "publish:cli": "npm run build && npm publish", "postinstall": "node hooks/postinstall.js", "prepublishOnly": "npm run build" diff --git a/cli/src/cli/utils/modelSelector.js b/cli/src/cli/utils/modelSelector.js index 438220d1..a2bd3fdc 100644 --- a/cli/src/cli/utils/modelSelector.js +++ b/cli/src/cli/utils/modelSelector.js @@ -55,7 +55,7 @@ async function getAvailableModelsGrouped() { } /** - * Display model list and prompt for selection + * Display model list and prompt for selection with provider grouping & search * @param {string} title - Title to display * @param {string} currentValue - Current selected value (optional) * @param {Object} options - { excludeCombos?: boolean } @@ -70,62 +70,199 @@ async function selectModelFromList(title, currentValue = "", options = {}) { if (totalModels === 0) { return null; } - - // Build flat list for selection - const allModels = []; - - // Display - clearScreen(); - console.log(`\n🎯 ${title}`); - console.log("=".repeat(50)); - if (currentValue) { - console.log(`Current: ${currentValue}\n`); - } else { - console.log(); - } - - let idx = 1; - - // Combos first (skipped when excludeCombos is true) + + // All models for flat search + const allModelsList = [ + ...combos, + ...Object.values(groups).flat() + ]; + + // Build category list + const categories = []; if (combos.length > 0) { - console.log("[Combos]"); - combos.forEach(combo => { - console.log(` ${idx}. ${combo}`); - allModels.push(combo); - idx++; + categories.push({ + id: "combos", + name: "[Combos]", + models: combos }); - console.log(); } - - // Provider groups in order (by alias) + const sortedProviders = Object.keys(groups).sort((a, b) => { const idxA = PROVIDER_ALIAS_ORDER.indexOf(a); const idxB = PROVIDER_ALIAS_ORDER.indexOf(b); return (idxA === -1 ? 999 : idxA) - (idxB === -1 ? 999 : idxB); }); - - sortedProviders.forEach(provider => { + + sortedProviders.forEach((provider) => { const providerName = PROVIDER_ALIAS_NAMES[provider] || provider; - console.log(`[${providerName}]`); - groups[provider].forEach(model => { - console.log(` ${idx}. ${model}`); - allModels.push(model); - idx++; + categories.push({ + id: provider, + name: providerName, + models: groups[provider] }); - console.log(); }); - - console.log(" 0. Cancel\n"); - - // Prompt for number input - const input = await prompt("Enter number: "); - const num = parseInt(input, 10); - - if (isNaN(num) || num === 0 || num < 0 || num > allModels.length) { - return null; + + let filterQuery = null; + + while (true) { + clearScreen(); + console.log(`\n🎯 ${title}`); + console.log("=".repeat(50)); + if (currentValue) { + console.log(`Current: ${currentValue}\n`); + } else { + console.log(); + } + + // Active search view + if (filterQuery !== null) { + const q = filterQuery.toLowerCase().trim(); + const matched = allModelsList.filter((m) => m.toLowerCase().includes(q)); + + console.log(`🔍 Search results for "${filterQuery}": (${matched.length} found)\n`); + if (matched.length === 0) { + console.log(" No matching models found.\n"); + console.log(" 0. ← Back to providers"); + console.log(" s. Search again\n"); + const act = await prompt("Select option: "); + if (act.toLowerCase() === "s") { + const newQ = await prompt("Enter search keyword: "); + filterQuery = newQ.trim() || null; + } else { + filterQuery = null; + } + continue; + } + + matched.forEach((m, i) => { + console.log(` ${i + 1}. ${m}`); + }); + console.log("\n 0. ← Back to providers"); + console.log(" s. Search again\n"); + + const input = await prompt("Enter number to select (or 0/s): "); + if (input.toLowerCase() === "s") { + const newQ = await prompt("Enter search keyword: "); + filterQuery = newQ.trim() || null; + continue; + } + const num = parseInt(input, 10); + if (isNaN(num) || num === 0) { + filterQuery = null; + continue; + } + if (num > 0 && num <= matched.length) { + return matched[num - 1]; + } + continue; + } + + // If only 1 category exists, jump straight into its model list + if (categories.length === 1) { + const singleCategory = categories[0]; + console.log(`[${singleCategory.name}]`); + singleCategory.models.forEach((m, i) => { + console.log(` ${i + 1}. ${m}`); + }); + console.log(); + console.log(" s. 🔍 Search models"); + console.log(" m. ✍️ Enter custom model ID"); + console.log(" 0. Cancel\n"); + + const input = await prompt("Enter choice (number / s / m / 0): "); + const trimmed = input.trim(); + if (!trimmed || trimmed === "0") return null; + + const lower = trimmed.toLowerCase(); + if (lower === "s") { + const q = await prompt("Enter search keyword: "); + if (q.trim()) filterQuery = q.trim(); + continue; + } + if (lower === "m") { + const customModel = await prompt("Enter custom model ID: "); + if (customModel.trim()) return customModel.trim(); + continue; + } + + const num = parseInt(trimmed, 10); + if (!isNaN(num) && num > 0 && num <= singleCategory.models.length) { + return singleCategory.models[num - 1]; + } + filterQuery = trimmed; + continue; + } + + // Multiple categories view + console.log("[Providers & Groups]"); + categories.forEach((cat, i) => { + console.log(` ${i + 1}. ${cat.name} (${cat.models.length} models)`); + }); + + console.log(); + console.log(" s. 🔍 Search models"); + console.log(" m. ✍️ Enter custom model ID"); + console.log(" 0. Cancel\n"); + + const input = await prompt("Enter choice (number / keyword / s / m): "); + const trimmed = input.trim(); + + if (!trimmed || trimmed === "0") { + return null; + } + + const lower = trimmed.toLowerCase(); + if (lower === "s") { + const q = await prompt("Enter search keyword: "); + if (q.trim()) { + filterQuery = q.trim(); + } + continue; + } + + if (lower === "m") { + const customModel = await prompt("Enter custom model ID: "); + if (customModel.trim()) { + return customModel.trim(); + } + continue; + } + + const num = parseInt(trimmed, 10); + // Selected a category + if (!isNaN(num) && num > 0 && num <= categories.length) { + const selectedCategory = categories[num - 1]; + + while (true) { + clearScreen(); + console.log(`\n🎯 ${title} > ${selectedCategory.name}`); + console.log("=".repeat(50)); + if (currentValue) { + console.log(`Current: ${currentValue}\n`); + } else { + console.log(); + } + + selectedCategory.models.forEach((m, i) => { + console.log(` ${i + 1}. ${m}`); + }); + console.log("\n 0. ← Back\n"); + + const modelChoice = await prompt("Enter number to select (0 to back): "); + const modelNum = parseInt(modelChoice, 10); + if (isNaN(modelNum) || modelNum === 0) { + break; + } + if (modelNum > 0 && modelNum <= selectedCategory.models.length) { + return selectedCategory.models[modelNum - 1]; + } + } + continue; + } + + // User typed text directly -> treat as search query + filterQuery = trimmed; } - - return allModels[num - 1]; } module.exports = { diff --git a/docs/superpowers/plans/2026-09-04-opencode-go-session-header.md b/docs/superpowers/plans/2026-09-04-opencode-go-session-header.md new file mode 100644 index 00000000..546de027 --- /dev/null +++ b/docs/superpowers/plans/2026-09-04-opencode-go-session-header.md @@ -0,0 +1,261 @@ +# OpenCode Go Session Header Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Send a stable, conversation-scoped `x-opencode-session` header on every OpenCode Go request and install the patched CLI locally. + +**Architecture:** Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. `chatCore` passes the provider-scoped session resolved from the original request plus the detected client tool; the executor derives a request-local upstream session and delegates all existing transport, authentication, retry, and proxy behavior to `DefaultExecutor`. + +**Tech Stack:** Node.js ESM, Vitest, Next.js, npm CLI packaging, GitHub CLI. + +## Global Constraints + +- Apply the header to OpenCode Go chat completions, Claude Messages, and OpenAI Responses transports. +- Preserve a valid native `x-opencode-session`; hash all translated non-OpenCode identities to `ses_<32 lowercase hex>`. +- Namespace translated identities by detected client tool, using `generic` when unknown. +- Do not keep mutable per-request session state on the executor singleton or mutate the caller's credentials object. +- Do not change OpenCode Go models, routing, reasoning, tool behavior, dependencies, or unrelated providers. +- Reuse upstream issue #3759 instead of creating a duplicate issue. + +--- + +### Task 1: Add Failing OpenCode Go Session Tests + +**Files:** +- Create: `tests/unit/opencode-go-session.test.js` + +**Interfaces:** +- Consumes: `getExecutor(provider)` and `DefaultExecutor.buildHeaders(credentials, stream, url, model)`. +- Produces: the required public behavior for `OpenCodeGoExecutor.prepareRequestCredentials({ body, credentials, providerSessionId, clientTool })` and `OpenCodeGoExecutor.execute(args)`. + +- [ ] **Step 1: Write the failing tests** + +Create a Vitest suite that mocks `proxyAwareFetch`, obtains `getExecutor("opencode-go")`, and asserts: + +```js +const prepared = executor.prepareRequestCredentials({ + body: { messages: [{ role: "user", content: "hello" }] }, + credentials: { apiKey: "test-key", connectionId: "conn-a", rawHeaders: {} }, + providerSessionId: "conversation-a", + clientTool: "claude", +}); + +expect(prepared).not.toBe(credentials); +expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/); +expect(credentials).not.toHaveProperty("_opencodeGoSession"); +``` + +Cover native header preservation, stable values across all three runtime transports, different conversation IDs, different client tools using the same ID, connection fallback, no singleton state, no header on `DefaultExecutor("openai")`, and the final fetch headers returned by `execute()`. + +- [ ] **Step 2: Run the focused test and verify RED** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js +``` + +Expected: FAIL because `getExecutor("opencode-go")` still returns `DefaultExecutor` and `prepareRequestCredentials` does not exist. + +- [ ] **Step 3: Commit the failing test** + +```bash +git add tests/unit/opencode-go-session.test.js +git commit -m "test: cover OpenCode Go session headers" +``` + +### Task 2: Implement the Dedicated Executor + +**Files:** +- Create: `open-sse/executors/opencode-go.js` +- Modify: `open-sse/executors/index.js` + +**Interfaces:** +- Consumes: `DefaultExecutor`, `resolveSessionId()`, request `credentials.rawHeaders`, `providerSessionId`, and `clientTool`. +- Produces: `OpenCodeGoExecutor`, `prepareRequestCredentials()`, and an `execute()` override that delegates with cloned credentials. + +- [ ] **Step 1: Add the minimal executor implementation** + +Implement these rules: + +```js +function translatedSessionId(sessionId, clientTool) { + const digest = crypto + .createHash("sha256") + .update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`) + .digest("hex") + .slice(0, 32); + return `ses_${digest}`; +} +``` + +`prepareRequestCredentials()` must read a case-insensitive native +`x-opencode-session` with the same non-empty, 256-character cap used by the +session manager. Otherwise it uses `providerSessionId` or calls +`resolveSessionId({ headers, body, connectionId, scope: "opencode-go" })`, then +returns `{ ...credentials, _opencodeGoSession: value }`. + +`execute(args)` must call `prepareRequestCredentials(args)` and delegate using +`super.execute({ ...args, credentials: prepared })`. `buildHeaders()` must call +`super.buildHeaders()` and add the prepared session, with a connection-scoped +fallback for direct callers. + +Register `new OpenCodeGoExecutor()` under `"opencode-go"` and export the class. + +- [ ] **Step 2: Run the focused test and verify partial GREEN** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js +``` + +Expected: executor-level tests pass; any chatCore-context assertion remains failing until Task 3. + +- [ ] **Step 3: Commit the executor** + +```bash +git add open-sse/executors/opencode-go.js open-sse/executors/index.js tests/unit/opencode-go-session.test.js +git commit -m "fix(opencode-go): add stable session header executor" +``` + +### Task 3: Pass Original Request Session Context + +**Files:** +- Modify: `open-sse/handlers/chatCore.js` +- Modify: `tests/unit/opencode-go-session.test.js` + +**Interfaces:** +- Consumes: existing `sessionSeed` and `clientTool` variables in `handleChatCore()`. +- Produces: `providerSessionId` and `clientTool` fields on both initial and refreshed-credential calls to `executor.execute()`. + +- [ ] **Step 1: Add or enable the failing integration assertion** + +Use a mocked executor or source request containing a body-only `session_id` and +assert the executor receives the provider-scoped session resolved before +translation. + +- [ ] **Step 2: Run the focused test and verify RED** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js tests/unit/opencode-go-session.test.js +``` + +Expected: FAIL because `handleChatCore()` does not pass `providerSessionId` or +`clientTool` to `executor.execute()`. + +- [ ] **Step 3: Pass the request context** + +Add the same fields to both executor calls: + +```js +executor.execute({ + model, + body: translatedBody, + stream, + credentials, + providerSessionId: sessionSeed, + clientTool, + signal: streamController.signal, + log, + proxyOptions, +}); +``` + +- [ ] **Step 4: Run focused and neighboring tests** + +Run: + +```bash +npx vitest run --config tests/vitest.config.js \ + tests/unit/opencode-go-session.test.js \ + tests/unit/opencode-go-models.test.js \ + tests/unit/session-manager.test.js \ + tests/unit/executor-const-guard.test.js +``` + +Expected: PASS with zero failed tests. + +- [ ] **Step 5: Commit the context wiring** + +```bash +git add open-sse/handlers/chatCore.js tests/unit/opencode-go-session.test.js +git commit -m "fix(chat): forward provider session context" +``` + +### Task 4: Verify and Install the Local CLI Package + +**Files:** +- Generated: `9router-0.5.65.tgz` +- Packaged output: `cli/app/server.js` + +**Interfaces:** +- Consumes: completed source changes and existing CLI build scripts. +- Produces: a globally installed patched `9router@0.5.65`. + +- [ ] **Step 1: Run source verification** + +```bash +git diff --check origin/master...HEAD +npx vitest run --config tests/vitest.config.js tests/unit/ +npm run build +``` + +Expected: every command exits zero. Record any pre-existing full-suite failures +separately rather than hiding them. + +- [ ] **Step 2: Build and package the CLI** + +```bash +npm --prefix cli run build +npm --prefix cli pack -- --pack-destination .. +``` + +Expected: `9router-0.5.65.tgz` exists and contains the patched bundled server. + +- [ ] **Step 3: Replace the global npm installation** + +```bash +npm install -g ./9router-0.5.65.tgz +``` + +Expected: `/opt/homebrew/lib/node_modules/9router/package.json` reports `0.5.65` +and the installed bundle contains `x-opencode-session` plus the new executor. + +- [ ] **Step 4: Commit any required package-source adjustment** + +Do not commit generated tarballs or CLI build artifacts unless the repository +already tracks and requires them. + +### Task 5: Publish the Upstream Pull Request + +**Files:** +- No additional source files unless verification finds a required correction. + +**Interfaces:** +- Consumes: verified branch commits and GitHub issue #3759. +- Produces: a fork branch and a PR against `decolua/9router:master`. + +- [ ] **Step 1: Create or repair the GitHub fork remote** + +Use `gh repo fork decolua/9router --remote` if the current `fork` remote remains +missing, then push `fix/opencode-go-session-header`. + +- [ ] **Step 2: Create the PR** + +Use title: + +```text +fix(opencode-go): send stable session header +``` + +The body must include the root cause, downstream-session translation policy, +three covered transports, concurrency behavior, verification evidence, +`Fixes #3759`, and a note that this PR is intentionally narrower than #3780. + +- [ ] **Step 3: Verify the published PR** + +Run `gh pr view --json number,title,state,url,headRefName,baseRefName` and report +the issue and PR URLs. diff --git a/docs/superpowers/specs/2026-09-04-opencode-go-session-header-design.md b/docs/superpowers/specs/2026-09-04-opencode-go-session-header-design.md new file mode 100644 index 00000000..94daf644 --- /dev/null +++ b/docs/superpowers/specs/2026-09-04-opencode-go-session-header-design.md @@ -0,0 +1,114 @@ +# OpenCode Go Session Header Design + +## Problem + +OpenCode Go will begin rejecting some requests without an +`x-opencode-session` header on September 6, 2026. In 9Router v0.5.65, +`opencode-go` uses `DefaultExecutor`, whose generic header builder does not add +that header. The specialized OpenCode Free executor already sends it, but that +logic does not apply to the paid OpenCode Go provider or its three transports. + +## Goals + +- Add `x-opencode-session` to every OpenCode Go chat, Claude Messages, and + OpenAI Responses request. +- Translate a downstream conversation identity into a stable upstream identity. +- Keep identities isolated across different downstream agents and conversations. +- Avoid exposing non-OpenCode downstream session identifiers to OpenCode Go. +- Avoid mutable session state on the shared executor singleton. +- Leave OpenCode Free and all unrelated providers unchanged. + +## Non-Goals + +- Inferring an exact conversation boundary when a downstream client provides no + session or conversation identifier. +- Adding or changing OpenCode Go models, routing, reasoning, or tool behavior. +- Changing the general session-resolution policy for other providers. + +## Architecture + +Add a dedicated `OpenCodeGoExecutor` extending `DefaultExecutor`. The executor +keeps the existing generic URL, authentication, translation, retry, and proxy +behavior, and overrides only the OpenCode Go session-header concern. + +`handleChatCore` already resolves a provider-scoped session from the original +request before translation. It will pass that value and the detected client +tool to `executor.execute()` as request context. `OpenCodeGoExecutor.execute()` +will create a shallow request-local credentials object containing the resolved +OpenCode Go session. It will then delegate to `DefaultExecutor.execute()`. +This avoids storing request state on the executor singleton or mutating shared +provider credentials. + +## Session Resolution + +The original downstream request remains the source of truth. Existing +`resolveSessionId()` behavior recognizes Claude Code, Antigravity, generic +session headers, and common body fields before request translation can discard +them. + +Resolution rules: + +1. If the downstream request supplies `x-opencode-session`, treat it as an + authoritative OpenCode identity after trimming and length validation. +2. Otherwise use the provider-scoped session resolved from the original request. +3. Namespace the resolved value with the detected downstream agent, falling back + to `generic` when the agent is unknown. +4. Convert the namespaced value to an opaque deterministic identifier: + `ses_` plus the first 32 hexadecimal characters of SHA-256. +5. If no explicit downstream identity exists, the existing provider connection + fallback guarantees that a header is still sent. It is stable but cannot + distinguish multiple conversations sharing that connection. + +The same input conversation produces the same upstream identifier for all three +OpenCode Go transports. Different agents using the same raw session value +produce different identifiers. + +## Header Injection + +`OpenCodeGoExecutor.buildHeaders()` delegates to +`DefaultExecutor.buildHeaders()` and adds only: + +```text +x-opencode-session: +``` + +The implementation applies to: + +- `https://opencode.ai/zen/go/v1/chat/completions` +- `https://opencode.ai/zen/go/v1/messages` +- `https://opencode.ai/zen/go/v1/responses` + +## Error Handling + +Session derivation must not make requests fail. Invalid or oversized native +header values are ignored and the normal resolved-session fallback is used. +Hashing uses Node's built-in `crypto` module and requires no new dependency. + +## Testing + +Add a focused unit suite that proves: + +- all three OpenCode Go transports receive the header; +- the same conversation remains stable across requests and transports; +- different conversations produce different values; +- different agents using the same raw ID remain isolated; +- non-OpenCode session IDs are represented as opaque `ses_<32 hex>` values; +- a valid native `x-opencode-session` remains stable; +- headerless requests still receive a stable fallback; +- OpenCode Free behavior is unchanged; +- unrelated `DefaultExecutor` providers do not receive the header; +- no request state is retained on the shared executor instance. + +Run the focused unit tests first, then the neighboring executor/session tests, +the full offline test suite, the application build, and the CLI package build. + +## Delivery + +Build the CLI with `npm --prefix cli run build`, create a package with +`npm --prefix cli pack`, and install the generated tarball globally to replace +the current npm-installed `9router@0.5.65`. Verify the installed package version +and packaged source contains the new executor. + +Upstream issue #3759 already tracks the problem, so no duplicate issue will be +created. The pull request will be narrowly scoped to this fix, reference +`Fixes #3759`, and explain how it differs from the broader open PR #3780. diff --git a/gitbook/content/en/providers/subscription.md b/gitbook/content/en/providers/subscription.md index 830f429e..ffe5fb0d 100644 --- a/gitbook/content/en/providers/subscription.md +++ b/gitbook/content/en/providers/subscription.md @@ -111,6 +111,27 @@ Model: cx/gpt-5.2-codex | `cx/gpt-5.2` | GPT 5.2 | General tasks | | `cx/gpt-5.1-codex` | GPT 5.1 Codex | Stable coding | +### Image Generation + +The Codex image catalog includes `cx/gpt-5.6-sol-image`, +`cx/gpt-5.6-terra-image`, and `cx/gpt-5.6-luna-image`, alongside the existing +GPT 5.5, 5.4, and 5.3 image aliases. Select them under **Image → OpenAI Codex** +in the dashboard, or discover them with `GET /v1/models/image` after connecting +a Codex account. + +```bash +curl http://localhost:20128/v1/images/generations \ + -H "Authorization: Bearer $NINE_ROUTER_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"model":"cx/gpt-5.6-sol-image","prompt":"A blue square","size":"1024x1024"}' +``` + +These are 9Router aliases: the image adapter removes `-image` and sends the +underlying model an `image_generation` tool through the Codex Responses API. +The same endpoint accepts an `image` reference for edits. Image generation +requires an eligible ChatGPT Plus or higher account; availability of each +underlying model and its image tool depends on the connected account. + ### Pro Tips - **5-hour rolling quota** - Fresh quota every 5 hours diff --git a/open-sse/config/appConstants.js b/open-sse/config/appConstants.js index 3e18633e..62ddfb6b 100644 --- a/open-sse/config/appConstants.js +++ b/open-sse/config/appConstants.js @@ -7,6 +7,9 @@ import { createRequire } from "module"; export const GEMINI_CLI_VERSION = PROVIDERS["gemini-cli"]?.cliVersion; export const GEMINI_CLI_API_CLIENT = PROVIDERS["gemini-cli"]?.apiClient; +// === Codex CLI === derive từ registry codex.transport +export const CODEX_CLI_VERSION = PROVIDERS["codex"]?.cliVersion; + // Map Node arch to Gemini CLI arch string (x64/x86/arm64/...) function geminiCLIArch() { const a = arch(); @@ -175,6 +178,11 @@ export const CLAUDE_SYSTEM_PROMPT = "You are Claude Code, Anthropic's official C // makes the backend flag the request and answer 429 Quota Exhausted. export const ANTIGRAVITY_PROMPT_REWRITES = [ { from: "You are a Claude agent, built on Anthropic's Claude Agent SDK.", to: "" }, + { from: /You are Hermes Agent,\s*(an intelligent AI assistant)(?: created by Nous Research)?\./gi, to: "You are Hermes Agent. You are $1." }, + // Claude Code prepends this line to its system prompt. The Claude-format translator strips it, + // but OpenAI-format clients (e.g. proxies that convert Claude Code to /v1/chat/completions) + // pass it through, and any system text containing it gets a fake 429 RESOURCE_EXHAUSTED. + { from: /^x-anthropic-billing-header:[^\n]*(?:\r?\n)*/gim, to: "" }, { from: /opencode/gi, to: (m) => (m === "OpenCode" ? "Antigravity" : m === "OPENCODE" ? "ANTIGRAVITY" : "antigravity") } ]; diff --git a/open-sse/config/providerModels.js b/open-sse/config/providerModels.js index 90a6b0b2..afa98fa7 100644 --- a/open-sse/config/providerModels.js +++ b/open-sse/config/providerModels.js @@ -27,11 +27,14 @@ const DOT_VERSION_PROVIDERS = new Set(["kr", "kiro"]); // ("claude-sonnet-4-5" ~= "claude-sonnet-4.5"). Other providers use exact match only. function findModel(models, modelId, aliasOrId) { if (!models) return undefined; - const found = models.find(m => m.id === modelId); + const baseModelId = typeof modelId === "string" + ? modelId.replace(/\([^()]+\)\s*$/, "").trim() + : modelId; + const found = models.find(m => m.id === modelId || m.id === baseModelId); if (found) return found; if (!DOT_VERSION_PROVIDERS.has(aliasOrId)) return undefined; - const normalized = normalizeModelId(modelId); - if (normalized === modelId) return undefined; + const normalized = normalizeModelId(baseModelId); + if (normalized === baseModelId) return undefined; return models.find(m => m.id === normalized); } @@ -50,7 +53,7 @@ export function findModelName(aliasOrId, modelId) { } export function getModelTargetFormat(aliasOrId, modelId) { - if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode") && isMuseSparkModel(modelId)) { + if ((!aliasOrId || aliasOrId === "oc" || aliasOrId === "opencode" || aliasOrId === "ocg" || aliasOrId === "opencode-go") && isMuseSparkModel(modelId)) { return FORMATS.OPENAI_RESPONSES; } const models = PROVIDER_MODELS[aliasOrId]; diff --git a/open-sse/executors/antigravity.js b/open-sse/executors/antigravity.js index 35ec1006..da4a7981 100644 --- a/open-sse/executors/antigravity.js +++ b/open-sse/executors/antigravity.js @@ -3,10 +3,11 @@ import { BaseExecutor } from "./base.js"; import { PROVIDERS } from "../config/providers.js"; import { OAUTH_ENDPOINTS, ANTIGRAVITY_HEADERS, AG_DEFAULT_TOOLS, AG_TOOL_SUFFIX, ANTIGRAVITY_PROMPT_REWRITES } from "../config/appConstants.js"; import { HTTP_STATUS } from "../config/runtimeConfig.js"; -import { resolveSessionId } from "../utils/sessionManager.js"; +import { resolveSessionId, toNumericSessionId } from "../utils/sessionManager.js"; import { proxyAwareFetch } from "../utils/proxyFetch.js"; -import { cleanJSONSchemaForAntigravity } from "../translator/formats/gemini.js"; +import { cleanJSONSchemaForAntigravity, normalizeGeminiContents } from "../translator/formats/gemini.js"; import { DEFAULT_THINKING_AG_SIGNATURE } from "../config/defaultThinkingSignature.js"; +import { getGeminiThoughtSignatureSync } from "../services/thoughtSignatureStore.js"; // Sanitize function name: Gemini requires [a-zA-Z_][a-zA-Z0-9_.:\-]{0,63} function sanitizeFunctionName(name) { @@ -187,9 +188,12 @@ export class AntigravityExecutor extends BaseExecutor { }; } + const rawSessionId = body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }); + const sessionId = toNumericSessionId(rawSessionId) || rawSessionId; + // ─── Standard (non-image) request ─── // Fix contents for Claude models via Antigravity - const contents = body.request?.contents?.map(c => { + const rawContents = (body.request?.contents || []).map(c => { let role = c.role; // functionResponse must be role "user" for Claude models if (c.parts?.some(p => p.functionResponse)) { @@ -202,21 +206,33 @@ export class AntigravityExecutor extends BaseExecutor { return true; }); // Gemini 3+ rejects functionCall parts without thoughtSignature. Clients (Claude Code, IDE) - // don't persist thoughtSignature in their history, so backfill the default signature on any - // functionCall part that arrives without one. - const needsBackfill = parts?.some(p => p.functionCall && !p.thoughtSignature) ?? false; - if (role !== c.role || parts?.length !== c.parts?.length || needsBackfill) { - return { - ...c, role, - parts: needsBackfill - ? parts.map(p => (p.functionCall && !p.thoughtSignature) - ? { ...p, thoughtSignature: DEFAULT_THINKING_AG_SIGNATURE } - : p) - : parts, - }; - } - return c; + // don't persist thoughtSignature in their history, so backfill from cache or default signature. + // In parallel function calls, only the first call needs a signature; siblings stay unsigned. + let firstFunctionCallSeen = false; + const modifiedParts = parts?.map(p => { + if (!p.functionCall) return p; + const callId = p.functionCall.id; + const cachedSig = callId ? getGeminiThoughtSignatureSync(callId, sessionId, body.model || model) : null; + const callSig = p.thoughtSignature || cachedSig || (!firstFunctionCallSeen ? DEFAULT_THINKING_AG_SIGNATURE : undefined); + firstFunctionCallSeen = true; + if (callSig) { + return { ...p, thoughtSignature: callSig }; + } + if (p.thoughtSignature && !cachedSig) { + // Unsigned sibling call + const { thoughtSignature: _, ...rest } = p; + return rest; + } + return p; + }); + + return { + ...c, + role, + parts: modifiedParts || parts || [], + }; }); + const contents = normalizeGeminiContents(rawContents); // Sanitize tool schemas and function names before sending to Antigravity. let tools = body.request?.tools; @@ -267,7 +283,7 @@ export class AntigravityExecutor extends BaseExecutor { generationConfig, ...(contents && { contents }), ...(tools && { tools }), - sessionId: body.request?.sessionId || resolveSessionId({ headers: credentials?.rawHeaders, body, connectionId: credentials?.email || credentials?.connectionId, scope: "antigravity" }), + sessionId, safetySettings: undefined, ...(tools?.length > 0 && { toolConfig: { functionCallingConfig: { mode: "VALIDATED" } } }) }; diff --git a/open-sse/executors/base.js b/open-sse/executors/base.js index 7dbb0700..e469e407 100644 --- a/open-sse/executors/base.js +++ b/open-sse/executors/base.js @@ -127,7 +127,7 @@ export class BaseExecutor { for (let urlIndex = 0; urlIndex < fallbackCount; urlIndex++) { const url = this.buildUrl(model, stream, urlIndex, credentials); const transformedBody = this.transformRequest(model, body, stream, credentials); - const headers = this.buildHeaders(credentials, stream, url, model); + const headers = this.buildHeaders(credentials, stream, url, model, transformedBody); if (!retryAttemptsByUrl[urlIndex]) retryAttemptsByUrl[urlIndex] = 0; diff --git a/open-sse/executors/codex.js b/open-sse/executors/codex.js index 4d9acbd2..de2af822 100644 --- a/open-sse/executors/codex.js +++ b/open-sse/executors/codex.js @@ -12,6 +12,7 @@ import { getThinkingLevels } from "../providers/thinkingLevels.js"; import { DEFAULT_RETRY_CONFIG, HTTP_STATUS, resolveRetryEntry } from "../config/runtimeConfig.js"; import { dbg } from "../utils/debugLog.js"; import { resolveSessionId } from "../utils/sessionManager.js"; +import { stripCodexUnsupportedPatterns } from "../utils/codexToolSchema.js"; // SSE error patterns inside 200-OK bodies. Some retry same account first; capacity rotates accounts. const CODEX_SSE_RETRY_PATTERNS = ["server_is_overloaded", "service_unavailable_error"]; @@ -72,6 +73,9 @@ function stripStoredItemReferences(body) { function normalizeCodexTools(body) { if (!Array.isArray(body.tools)) return; const validNames = new Set(); + // Codex's schema validator has no Unicode property escapes; a `pattern` + // carrying `\p{...}` 400s the whole request on every account (#3922). + const patternStats = { removed: 0 }; body.tools = body.tools.filter((tool) => { if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false; const type = typeof tool.type === "string" ? tool.type : ""; @@ -80,6 +84,9 @@ function normalizeCodexTools(body) { for (const st of tool.tools) { const n = typeof st?.name === "string" ? st.name.trim().slice(0, 128) : ""; if (n) validNames.add(n); + if (st?.parameters && typeof st.parameters === "object") { + st.parameters = stripCodexUnsupportedPatterns(st.parameters, patternStats); + } } } return true; @@ -101,10 +108,13 @@ function normalizeCodexTools(body) { tool.type = "function"; tool.name = name.slice(0, 128); if (description) tool.description = description; - tool.parameters = parameters; + tool.parameters = stripCodexUnsupportedPatterns(parameters, patternStats); validNames.add(name); return true; }); + if (patternStats.removed > 0) { + dbg("CODEX", `stripped ${patternStats.removed} unsupported tool schema pattern(s)`); + } // Drop tool_choice if it references an unknown function name if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) { if (body.tool_choice.type === "function") { diff --git a/open-sse/executors/commandcode.js b/open-sse/executors/commandcode.js index f694e61b..d923fd50 100644 --- a/open-sse/executors/commandcode.js +++ b/open-sse/executors/commandcode.js @@ -40,10 +40,24 @@ export class CommandCodeExecutor extends BaseExecutor { } async execute(opts) { - const result = await super.execute(opts); - if (!result?.response?.ok || !result.response.body) return result; - result.response = await inspectAndWrapCommandCodeResponse(result.response, opts.model); - return result; + const maxRetries = 2; + for (let attempt = 0; attempt <= maxRetries; attempt++) { + const result = await super.execute(opts); + if (!result?.response?.ok || !result.response.body) return result; + + const wrappedResponse = await inspectAndWrapCommandCodeResponse(result.response, opts.model); + if (!wrappedResponse.ok && attempt < maxRetries) { + const isRetryableStatus = wrappedResponse.status === 502 || wrappedResponse.status === 503 || wrappedResponse.status === 504; + if (isRetryableStatus) { + opts.log?.debug?.("RETRY", `CommandCode upstream returned status ${wrappedResponse.status}, retrying ${attempt + 1}/${maxRetries}...`); + await new Promise(r => setTimeout(r, 1000 * (attempt + 1))); + continue; + } + } + + result.response = wrappedResponse; + return result; + } } parseError(response, bodyText) { diff --git a/open-sse/executors/default.js b/open-sse/executors/default.js index dcb98eb1..50e08936 100644 --- a/open-sse/executors/default.js +++ b/open-sse/executors/default.js @@ -151,7 +151,7 @@ export class DefaultExecutor extends BaseExecutor { return BEARER; } - buildHeaders(credentials, stream = true, url, model) { + buildHeaders(credentials, stream = true, url, model, body = null) { const rt = credentials?.runtimeTransport; const headers = { "Content-Type": "application/json", ...(rt ? rt.headers : this.config.headers) }; const desc = rt?.auth || AUTH_DESCRIPTORS[this.provider] || this.resolveAuthDescriptor(); @@ -159,8 +159,19 @@ export class DefaultExecutor extends BaseExecutor { for (const hook of desc.hooks || []) HEADER_HOOKS[hook]?.(headers, credentials); applyAuth(headers, desc, credentials); - if (this.provider === "claude" && model) { - headers["Anthropic-Beta"] = selectAnthropicBeta(model); + // anthropic-compatible-* nodes serving a real Claude model sit in front of + // Anthropic itself (a rotating multi-account proxy, a corporate gateway), + // so the request needs the same beta flags the `claude` provider sends: + // without `context-management-2025-06-27` upstream rejects the + // `context_management` block Claude Code puts in every request with + // "context_management: Extra inputs are not permitted" (HTTP 400), and the + // combo silently falls through to the next model. The model id gates this: + // a node fronting Kimi or GLM answers on its own ids and never matches, so + // gateways that would choke on unknown beta flags are left untouched. + const isClaudeModel = typeof model === "string" && /^claude-/.test(model); + if (model && (this.provider === "claude" + || (this.provider?.startsWith?.("anthropic-compatible-") && isClaudeModel))) { + headers["Anthropic-Beta"] = selectAnthropicBeta(model, body); } // Strip first-party Claude Code identity headers for non-Anthropic anthropic-compatible upstreams diff --git a/open-sse/executors/index.js b/open-sse/executors/index.js index a2d0fd68..157af0b9 100644 --- a/open-sse/executors/index.js +++ b/open-sse/executors/index.js @@ -10,12 +10,14 @@ import { CodexExecutor } from "./codex.js"; import { CursorExecutor } from "./cursor.js"; import { VertexExecutor } from "./vertex.js"; import { OpenCodeExecutor } from "./opencode.js"; +import { OpenCodeGoExecutor } from "./opencode-go.js"; import { GrokWebExecutor } from "./grok-web.js"; import { GrokCliExecutor } from "./grok-cli.js"; import { PerplexityWebExecutor } from "./perplexity-web.js"; import { OllamaLocalExecutor } from "./ollama-local.js"; import { CommandCodeExecutor } from "./commandcode.js"; import { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js"; +import { XiaomiMimoExecutor } from "./xiaomi-mimo.js"; import { MimoFreeExecutor } from "./mimo-free.js"; import { CodeBuddyExecutor } from "./codebuddy-cn.js"; import { CodeBuddyIntlExecutor } from "./codebuddy-intl.js"; @@ -41,6 +43,7 @@ const executors = { vertex: new VertexExecutor("vertex"), "vertex-partner": new VertexExecutor("vertex-partner"), opencode: new OpenCodeExecutor(), + "opencode-go": new OpenCodeGoExecutor(), "grok-web": new GrokWebExecutor(), "grok-cli": new GrokCliExecutor(), gcli: new GrokCliExecutor(), // Alias @@ -49,6 +52,7 @@ const executors = { "ollama-local": new OllamaLocalExecutor(), commandcode: new CommandCodeExecutor(), "xiaomi-tokenplan": new XiaomiTokenplanExecutor(), + "xiaomi-mimo": new XiaomiMimoExecutor(), "mimo-free": new MimoFreeExecutor(), mmf: new MimoFreeExecutor(), // Alias for mimo-free "codebuddy-cn": new CodeBuddyExecutor(), @@ -86,12 +90,14 @@ export { CursorExecutor } from "./cursor.js"; export { VertexExecutor } from "./vertex.js"; export { DefaultExecutor } from "./default.js"; export { OpenCodeExecutor } from "./opencode.js"; +export { OpenCodeGoExecutor } from "./opencode-go.js"; export { GrokWebExecutor } from "./grok-web.js"; export { GrokCliExecutor } from "./grok-cli.js"; export { PerplexityWebExecutor } from "./perplexity-web.js"; export { OllamaLocalExecutor } from "./ollama-local.js"; export { CommandCodeExecutor } from "./commandcode.js"; export { XiaomiTokenplanExecutor } from "./xiaomi-tokenplan.js"; +export { XiaomiMimoExecutor } from "./xiaomi-mimo.js"; export { MimoFreeExecutor } from "./mimo-free.js"; export { CodeBuddyExecutor } from "./codebuddy-cn.js"; export { CodeBuddyIntlExecutor } from "./codebuddy-intl.js"; diff --git a/open-sse/executors/kiro.js b/open-sse/executors/kiro.js index 77616618..9d5d754f 100644 --- a/open-sse/executors/kiro.js +++ b/open-sse/executors/kiro.js @@ -127,12 +127,18 @@ async function readResponsePrefix(response, signal, maxBytes, timeoutMs) { return decoder.decode(concatChunks(chunks, totalBytes)); } +// The instruction goes into the current user turn, never into a top-level +// `systemPrompt`: kiro.dev answers any body carrying that field with +// 400 REQUEST_BODY_INVALID, so writing it here turned every repair retry into +// a hard failure. function appendRepairInstruction(body, kind) { const repaired = structuredClone(body || {}); const instruction = REPAIR_INSTRUCTIONS[kind] || "Retry the previous incomplete Kiro response."; - repaired.systemPrompt = repaired.systemPrompt - ? `${repaired.systemPrompt}\n\n${instruction}` - : instruction; + const msg = repaired?.conversationState?.currentMessage?.userInputMessage; + if (msg) { + const content = typeof msg.content === "string" ? msg.content : ""; + msg.content = content ? `${content}\n\n${instruction}` : instruction; + } return repaired; } @@ -259,6 +265,19 @@ export class KiroExecutor extends BaseExecutor { } } + // CLIRO parity for the Amazon surfaces: the Kiro runtime accepts the + // SSO bearer header + agent-mode marker. Without these the deprecated + // path gateway answers REQUEST_BODY_INVALID for modern payloads. + if (credentials?.accessToken) { + headers["x-amz-sso-bearer"] = credentials.accessToken; + } + headers["x-amzn-kiro-agent-mode"] = "spec"; + headers["x-amzn-codewhisperer-machine-id"] = "kiro-desktop"; + const profileArn = credentials?.providerSpecificData?.profileArn; + if (profileArn) { + headers["x-amzn-codewhisperer-profile-arn"] = profileArn; + } + return headers; } @@ -285,9 +304,13 @@ export class KiroExecutor extends BaseExecutor { // 403 "bearer token invalid", so they must hit the CodeWhisperer // *.amazonaws.com surface, and in the region the token was minted in // (the baseUrls are hardcoded us-east-1). - const isCodeWhispererSurface = - authMethod === "api_key" || authMethod === "external_idp" || authMethod === "idc"; - if (!isCodeWhispererSurface) return baseUrls; + // Kiro deprecated the legacy path-style GenerateAssistantResponse on + // runtime.*.kiro.dev (IDE 1.0.228+ moved to POST / + x-amz-target). The + // path gateway now answers valid modern payloads with 400 + // REQUEST_BODY_INVALID, and 400 is terminal in BaseExecutor, so kiro.dev + // must never be the first surface for any auth method. Amazon surfaces + // reject foreign tokens with 401/403, which DO fall through, so trying + // q/codewhisperer first is safe for every auth method (CLIRO parity). const region = (credentials?.providerSpecificData?.region || "us-east-1").trim(); const regionalize = (u) => @@ -297,20 +320,17 @@ export class KiroExecutor extends BaseExecutor { const amazon = baseUrls.filter((u) => u.includes("amazonaws.com")).map(regionalize); const others = baseUrls.filter((u) => !u.includes("amazonaws.com")); - if (authMethod === "api_key") { - const q = amazon.filter((u) => u.includes("://q.")); - const remaining = amazon.filter((u) => !u.includes("://q.")); - return q.length > 0 - ? [...q, ...remaining, ...others] - : [...amazon, ...others]; - } - - return amazon.length > 0 ? [...amazon, ...others] : baseUrls; + const q = amazon.filter((u) => u.includes("://q.")); + const remaining = amazon.filter((u) => !u.includes("://q.")); + return q.length > 0 + ? [...q, ...remaining, ...others] + : [...amazon, ...others]; } buildUrl(model, stream, urlIndex = 0, credentials = null) { const baseUrls = this.getOrderedBaseUrls(credentials); - return baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl; + const url = baseUrls[urlIndex] || baseUrls[0] || this.config.baseUrl; + return url; } // Retry only endpoint/auth-surface failures. Payload-invalid HTTP 400 must be diff --git a/open-sse/executors/opencode-go.js b/open-sse/executors/opencode-go.js new file mode 100644 index 00000000..fe90c6eb --- /dev/null +++ b/open-sse/executors/opencode-go.js @@ -0,0 +1,192 @@ +import crypto from "node:crypto"; +import { DefaultExecutor } from "./default.js"; +import { resolveSessionId } from "../utils/sessionManager.js"; +import { modelTargetFormat } from "../providers/models/schema.js"; +import { getProviderModels } from "../config/providerModels.js"; +import { + normalizeResponsesInput, + clampResponsesCallId, + coerceResponsesArguments, + coerceResponsesOutput, +} from "../translator/formats/responsesApi.js"; + +const SESSION_HEADER = "x-opencode-session"; +const SESSION_FIELD = "_opencodeGoSession"; +const MAX_SESSION_LENGTH = 256; + +const RESPONSES_BASE_URL = "https://opencode.ai/zen/go/v1/responses"; +const MAX_TOOL_NAME_LEN = 128; + +function normalizeSession(value) { + if (typeof value !== "string") return null; + const normalized = value.trim(); + if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null; + return normalized; +} + +function nativeSession(headers) { + if (!headers || typeof headers !== "object") return null; + for (const [key, value] of Object.entries(headers)) { + if (key.toLowerCase() === SESSION_HEADER) return normalizeSession(value); + } + return null; +} + +function translatedSession(sessionId, clientTool) { + const digest = crypto + .createHash("sha256") + .update(`opencode-go\0${clientTool || "generic"}\0${sessionId}`) + .digest("hex") + .slice(0, 32); + return `ses_${digest}`; +} + +// Strip the thinking suffix "model(level)" so checks hit the base id. +function baseModelId(model) { + return String(model || "").replace(/\([^()]+\)\s*$/, "").trim(); +} + +// Responses-only per the provider registry (grok-4.6, gpt-5.6-luna, muse-spark, …). +// Reading the registry keeps this in sync with config — never hardcode model ids here. +function isResponsesModel(model) { + const entry = getProviderModels("opencode-go").find((m) => m.id === baseModelId(model)); + return modelTargetFormat(entry) === "openai-responses"; +} + +// Flatten Chat Completions tool declarations into the Responses flat shape and +// drop hosted/nameless tools the /responses endpoint rejects. +function normalizeResponsesTools(body) { + if (!Array.isArray(body.tools)) return; + const validNames = new Set(); + body.tools = body.tools.filter((tool) => { + if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false; + const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null; + const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : ""); + const name = rawName.trim(); + if (!name) return false; + const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : ""); + let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)) + ? tool.parameters + : (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} }); + // Mirror the request translator: {type:"object"} without properties is rejected + // by strict Responses backends, so fill in the empty properties map. + if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} }; + for (const k of Object.keys(tool)) delete tool[k]; + tool.type = "function"; + tool.name = name.slice(0, MAX_TOOL_NAME_LEN); + if (description) tool.description = description; + tool.parameters = parameters; + validNames.add(tool.name); + return true; + }); + if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) { + if (body.tool_choice.type === "function") { + const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : ""; + if (!n || !validNames.has(n)) delete body.tool_choice; + } + } +} + +// Last line of defense for native Responses clients (sourceFormat === targetFormat +// skips translation): coerce items in place so malformed tool payloads 400 here +// with a clear shape instead of upstream as InputValidationError. +function sanitizeResponsesItems(body) { + if (!Array.isArray(body.input)) return; + body.input = body.input.filter((item) => { + if (!item || typeof item !== "object" || Array.isArray(item)) return true; + // Strip prior-turn reasoning items: Muse Spark contributor models route to + // an upstream Console backend where encrypted_content cannot be validated across + // rotated accounts or sessions, causing 400 "reasoning encrypted_content was not issued to this caller". + if (item.type === "reasoning") return false; + delete item.encrypted_content; + delete item.reasoning_encrypted_content; + if (item.type === "function_call") { + if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false; + item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN); + item.call_id = clampResponsesCallId(item.call_id); + item.arguments = coerceResponsesArguments(item.arguments); + return true; + } + if (item.type === "function_call_output") { + item.call_id = clampResponsesCallId(item.call_id); + item.output = coerceResponsesOutput(item.output); + return true; + } + return true; + }); +} + +export class OpenCodeGoExecutor extends DefaultExecutor { + constructor() { + super("opencode-go"); + } + + buildUrl(model, stream, urlIndex = 0, credentials = null) { + // Muse Spark lives on /responses even when a stale runtimeTransport leaks in. + if (isResponsesModel(model)) return RESPONSES_BASE_URL; + return super.buildUrl(model, stream, urlIndex, credentials); + } + + prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) { + const sourceCredentials = credentials || {}; + const native = nativeSession(sourceCredentials.rawHeaders); + const resolved = normalizeSession(providerSessionId) || resolveSessionId({ + headers: sourceCredentials.rawHeaders, + body, + connectionId: sourceCredentials.connectionId, + scope: "opencode-go", + }); + + return { + ...sourceCredentials, + [SESSION_FIELD]: native || translatedSession(resolved, clientTool), + }; + } + + async execute(args) { + const credentials = this.prepareRequestCredentials(args); + return super.execute({ ...args, credentials }); + } + + buildHeaders(credentials, stream = true, url, model) { + const headers = super.buildHeaders(credentials || {}, stream, url, model); + const prepared = credentials?.[SESSION_FIELD]; + if (prepared) { + headers[SESSION_HEADER] = prepared; + return headers; + } + + const fallback = this.prepareRequestCredentials({ credentials }); + headers[SESSION_HEADER] = fallback[SESSION_FIELD]; + return headers; + } + + transformRequest(model, body, stream, credentials) { + const out = super.transformRequest(model, body); + if (!isResponsesModel(model || body?.model)) return out; + const normalized = normalizeResponsesInput(out.input); + if (normalized) out.input = normalized; + if (!Array.isArray(out.input) || out.input.length === 0) { + out.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }]; + } + // Responses names the output cap max_output_tokens, not max_tokens. + if (out.max_output_tokens === undefined) { + if (out.max_completion_tokens !== undefined) out.max_output_tokens = out.max_completion_tokens; + else if (out.max_tokens !== undefined) out.max_output_tokens = out.max_tokens; + } + delete out.max_tokens; + delete out.max_completion_tokens; + if (out.reasoning_effort !== undefined && out.reasoning === undefined) { + out.reasoning = { effort: out.reasoning_effort, summary: "auto" }; + } + if (out.reasoning && typeof out.reasoning === "object" && !Array.isArray(out.reasoning)) { + if (!out.reasoning.summary) out.reasoning.summary = "auto"; + } + delete out.reasoning_effort; + out.stream = true; + out.store = false; + normalizeResponsesTools(out); + sanitizeResponsesItems(out); + return out; + } +} diff --git a/open-sse/executors/opencode.js b/open-sse/executors/opencode.js index ab14cb22..91abb3ac 100644 --- a/open-sse/executors/opencode.js +++ b/open-sse/executors/opencode.js @@ -1,24 +1,311 @@ import crypto from "crypto"; import { BaseExecutor } from "./base.js"; import { PROVIDERS } from "../config/providers.js"; +import { MEMORY_CONFIG } from "../config/runtimeConfig.js"; import { getThinkingLevels } from "../providers/thinkingLevels.js"; import { injectReasoningContent } from "../utils/reasoningContentInjector.js"; import { resolveSessionId } from "../utils/sessionManager.js"; import { isMuseSparkModel } from "../providers/models/helpers.js"; +import { ANTHROPIC_API_VERSION } from "../providers/shared.js"; +import { + normalizeResponsesInput, + clampResponsesCallId, + coerceResponsesArguments, + coerceResponsesOutput, +} from "../translator/formats/responsesApi.js"; -const OPENCODE_UA = "opencode"; +const OPENCODE_UA = "opencode/1.18.31"; +const MAX_SESSION_LENGTH = 256; +const MAX_TOOL_NAME_LEN = 128; +const SESSION_HEADER = "x-opencode-session"; +const SESSION_FIELD = "_opencodeSession"; +const REQ_FIELD = "_opencodeRequest"; +export const OPENCODE_SESSION_RE = /^ses_[0-9a-f]{12}[0-9A-Za-z]{14}$/; +export const OPENCODE_REQUEST_RE = /^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/; +const BASE62_CHARS = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"; + +// OpenCode free tier requires both 'bash' and 'read' in tools payload. +// Injected as cloaked decoy tools so external CLI tools (e.g. Claude Code's Bash/Read) +// take precedence while satisfying upstream verification. +const OPENCODE_DECOY_CHAT_TOOLS = [ + { + type: "function", + function: { + name: "bash", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, + }, + { + type: "function", + function: { + name: "read", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, + }, +]; + +const OPENCODE_DECOY_RESPONSES_TOOLS = [ + { + type: "function", + name: "bash", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, + { + type: "function", + name: "read", + description: "This tool is currently unavailable and must not be used.", + parameters: { type: "object", properties: {} }, + }, +]; + +function cloakOpencodeTools(body, isResponses) { + if (!body || typeof body !== "object") return; + if (isResponses) { + if (!Array.isArray(body.tools)) body.tools = []; + const names = new Set(body.tools.map((t) => t.name || t.function?.name)); + for (const tool of OPENCODE_DECOY_RESPONSES_TOOLS) { + if (!names.has(tool.name)) body.tools.push({ ...tool }); + } + if (!body.tool_choice) body.tool_choice = "auto"; + } else { + const hasTools = Array.isArray(body.tools) && body.tools.length > 0; + if (!hasTools) { + body.tools = OPENCODE_DECOY_CHAT_TOOLS.map((t) => ({ ...t, function: { ...t.function } })); + if (!body.tool_choice) body.tool_choice = "none"; + } else { + const names = new Set(body.tools.map((t) => t.function?.name || t.name)); + for (const tool of OPENCODE_DECOY_CHAT_TOOLS) { + if (!names.has(tool.function.name)) { + body.tools.push({ ...tool, function: { ...tool.function } }); + } + } + } + } +} + +function hasValidOpencodeVersion(ua) { + const m = String(ua || "").match(/opencode\/(\d+)\.(\d+)(?:\.(\d+))?/i); + if (!m) return false; + const major = parseInt(m[1], 10); + const minor = parseInt(m[2], 10); + return major > 1 || (major === 1 && minor >= 17); +} // Models served by /zen/v1/responses; every other model stays on /chat/completions. const RESPONSES_MODELS = new Set([ "muse-spark-1.2-contributor-free", "muse-spark-1.3-contributor-free", ]); +const MESSAGES_MODELS = new Set(["union-alpha"]); -function generateRequestId() { - return `msg_${crypto.randomUUID().replace(/-/g, "")}`; +let lastTimestamp = 0; +let counter = 0; + +function unstableRandom() { + const bytes = crypto.randomBytes(14); + let randomPart = ""; + for (let i = 0; i < 14; i++) { + randomPart += BASE62_CHARS[bytes[i] % 62]; + } + return randomPart; } -function generateSessionId() { - return `ses_${crypto.randomUUID().replace(/-/g, "")}`; +export function generateSessionId(timestamp = Date.now()) { + if (timestamp !== lastTimestamp) { + lastTimestamp = timestamp; + counter = 0; + } + counter++; + + const current = BigInt(timestamp) * 0x1000n + BigInt(counter); + const value = ~current; + const time = Array.from({ length: 6 }, (_, index) => + Number((value >> BigInt(40 - 8 * index)) & 0xffn) + .toString(16) + .padStart(2, "0") + ).join(""); + return `ses_${time}${unstableRandom()}`; +} + +export function generateRequestId(timestamp = Date.now()) { + const current = BigInt(timestamp) * 0x1000n + 1n; + const value = current; + const time = Array.from({ length: 6 }, (_, index) => + Number((value >> BigInt(40 - 8 * index)) & 0xffn) + .toString(16) + .padStart(2, "0") + ).join(""); + return `msg_${time}${unstableRandom()}`; +} + +export function translateSessionId(sessionId, clientTool = "") { + if (typeof sessionId === "string" && OPENCODE_SESSION_RE.test(sessionId.trim())) { + return sessionId.trim(); + } + const digest = crypto + .createHash("sha256") + .update(`opencode\0${clientTool || "generic"}\0${sessionId || ""}`) + .digest(); + const timeHex = digest.subarray(0, 6).toString("hex"); + let randomPart = ""; + for (let i = 6; i < 20; i++) { + randomPart += BASE62_CHARS[digest[i] % 62]; + } + return `ses_${timeHex}${randomPart}`; +} + +function normalizeSession(value) { + if (typeof value !== "string") return null; + const normalized = value.trim(); + if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null; + return normalized; +} + +function nativeSession(headers) { + if (!headers || typeof headers !== "object") return null; + for (const [key, value] of Object.entries(headers)) { + if (key.toLowerCase() === SESSION_HEADER) { + const normalized = normalizeSession(value); + if (normalized && OPENCODE_SESSION_RE.test(normalized)) return normalized; + } + } + return null; +} + +// Upstream free-tier quota is accounted per session. Minting a fresh +// x-opencode-session on every request burns through it and surfaces as +// 429 FreeUsageLimitError with growing reset-after delays, while the real +// CLI reuses one long-lived canonical session per conversation. Mirror +// that: one stable canonical session per downstream identity, evicted +// after MEMORY_CONFIG.sessionTtlMs like the other session stores. +const stableOpencodeSessions = new Map(); +const MAX_STABLE_SESSIONS = 1000; +const stableSessionCleanup = setInterval(() => { + const now = Date.now(); + for (const [key, entry] of stableOpencodeSessions) { + if (now - entry.lastUsed > MEMORY_CONFIG.sessionTtlMs) { + stableOpencodeSessions.delete(key); + } + } +}, MEMORY_CONFIG.sessionCleanupIntervalMs); +if (stableSessionCleanup.unref) stableSessionCleanup.unref(); + +function identityKey(credentials) { + const connectionId = credentials?.connectionId || credentials?.id; + if (connectionId) return `opencode:conn:${String(connectionId).slice(0, 128)}`; + const raw = credentials?.rawHeaders || {}; + const auth = raw.authorization || raw.Authorization || raw["x-api-key"] || raw["X-Api-Key"] || ""; + if (auth) { + const digest = crypto.createHash("sha256").update(String(auth)).digest("hex").slice(0, 32); + return `opencode:auth:${digest}`; + } + return "opencode:default"; +} + +export function stableSessionId(credentials) { + const key = identityKey(credentials); + const existing = stableOpencodeSessions.get(key); + if (existing) { + existing.lastUsed = Date.now(); + stableOpencodeSessions.delete(key); + stableOpencodeSessions.set(key, existing); + return existing.sessionId; + } + const sessionId = generateSessionId(); + if (stableOpencodeSessions.size >= MAX_STABLE_SESSIONS) { + stableOpencodeSessions.delete(stableOpencodeSessions.keys().next().value); + } + stableOpencodeSessions.set(key, { sessionId, lastUsed: Date.now() }); + return sessionId; +} + +function lastUserText(body) { + try { + if (!body || typeof body !== "object") return ""; + const arr = Array.isArray(body.messages) + ? body.messages + : Array.isArray(body.input) + ? body.input + : null; + if (!arr) return typeof body.input === "string" ? body.input.slice(-600) : ""; + for (let i = arr.length - 1; i >= 0; i--) { + const msg = arr[i]; + if (!msg) continue; + if (msg.role && msg.role !== "user") continue; + const content = msg.content; + if (typeof content === "string" && content.trim()) return content.trim().slice(-600); + if (Array.isArray(content)) { + const text = content + .map((part) => (typeof part === "string" ? part : part?.text || part?.input_text || "")) + .join(" ") + .trim(); + if (text) return text.slice(-600); + } + } + } catch { + return ""; + } + return ""; +} + +// The real CLI sends the current user message id (stable per turn, same on +// retries) as x-opencode-request. Derive it deterministically from the +// session plus the last user message so retries share the id. +export function deriveRequestId(sessionId, body) { + const text = lastUserText(body); + if (!text) return generateRequestId(); + const digest = crypto + .createHash("sha256") + .update(`opencode-req\0${sessionId || ""}\0${text}`) + .digest(); + const timeHex = digest.subarray(0, 6).toString("hex"); + let randomPart = ""; + for (let i = 6; i < 20; i++) { + randomPart += BASE62_CHARS[digest[i] % 62]; + } + const id = `msg_${timeHex}${randomPart}`; + return OPENCODE_REQUEST_RE.test(id) ? id : generateRequestId(); +} + +function normalizeRequestId(value) { + if (typeof value !== "string") return null; + const normalized = value.trim(); + if (!normalized || normalized.length > MAX_SESSION_LENGTH) return null; + return OPENCODE_REQUEST_RE.test(normalized) ? normalized : null; +} + +function bodyHasSessionHints(body) { + try { + if (!body || typeof body !== "object") return false; + if (typeof body.session_id === "string" && body.session_id.trim()) return true; + if (typeof body.conversation_id === "string" && body.conversation_id.trim()) return true; + if (typeof body.prompt_cache_key === "string" && body.prompt_cache_key.trim()) return true; + if (body.metadata && typeof body.metadata.user_id === "string" && body.metadata.user_id.trim()) return true; + if (body.request && body.request.sessionId != null && String(body.request.sessionId) !== "") return true; + const arr = Array.isArray(body.messages) + ? body.messages + : Array.isArray(body.input) + ? body.input + : null; + if (arr) { + let assistantText = ""; + for (const msg of arr) { + if (msg?.role === "assistant") { + const content = msg.content; + if (typeof content === "string") assistantText += content; + else if (Array.isArray(content)) { + for (const part of content) assistantText += part?.text || part?.output || ""; + } + if (assistantText.length >= 50) return true; + } + } + } + return false; + } catch { + return false; + } } // Strip the thinking suffix "model(level)" so registry lookups hit the base id. @@ -31,14 +318,116 @@ function isResponsesModel(model) { return RESPONSES_MODELS.has(base) || isMuseSparkModel(base); } -function resolveOpencodeSession(body, credentials) { +function isMessagesModel(model) { + return MESSAGES_MODELS.has(baseModelId(model)); +} + +function resolveOpencodeSession(body, credentials, providerSessionId, clientTool) { const headers = credentials?.rawHeaders || {}; - return resolveSessionId({ - headers, - body, - connectionId: credentials?.connectionId, - scope: "opencode", - generate: generateSessionId, + const native = nativeSession(headers); + if (native) return native; + + let incoming = null; + for (const [key, value] of Object.entries(headers)) { + if (key.toLowerCase() === SESSION_HEADER) { + incoming = normalizeSession(value); + break; + } + } + + const hinted = incoming || normalizeSession(providerSessionId); + if (hinted) return translateSessionId(hinted, clientTool); + + if (credentials?.connectionId || bodyHasSessionHints(body)) { + let viaManager = null; + try { + viaManager = resolveSessionId({ + headers, + body, + connectionId: credentials?.connectionId, + scope: "opencode", + }); + } catch { + viaManager = null; + } + if (viaManager) return translateSessionId(viaManager, clientTool); + } + + return stableSessionId(credentials); +} + +function resolveOpencodeRequestId(body, credentials, sessionId) { + const raw = credentials?.rawHeaders || {}; + for (const [key, value] of Object.entries(raw)) { + if (key.toLowerCase() === "x-opencode-request") { + const normalized = normalizeRequestId(value); + if (normalized) return normalized; + break; + } + } + return deriveRequestId(sessionId, body); +} + +function normalizeResponsesTools(body) { + if (!Array.isArray(body.tools)) return; + const validNames = new Set(); + body.tools = body.tools.filter((tool) => { + if (!tool || typeof tool !== "object" || Array.isArray(tool)) return false; + const fn = tool.function && typeof tool.function === "object" && !Array.isArray(tool.function) ? tool.function : null; + const rawName = typeof tool.name === "string" ? tool.name : (typeof fn?.name === "string" ? fn.name : ""); + const name = rawName.trim(); + if (!name) return false; + const description = typeof tool.description === "string" ? tool.description : (typeof fn?.description === "string" ? fn.description : ""); + let parameters = (tool.parameters && typeof tool.parameters === "object" && !Array.isArray(tool.parameters)) + ? tool.parameters + : (fn?.parameters && typeof fn.parameters === "object" && !Array.isArray(fn.parameters) ? fn.parameters : { type: "object", properties: {} }); + if (parameters.type === "object" && !parameters.properties) parameters = { ...parameters, properties: {} }; + for (const k of Object.keys(tool)) delete tool[k]; + tool.type = "function"; + tool.name = name.slice(0, MAX_TOOL_NAME_LEN); + if (description) tool.description = description; + tool.parameters = parameters; + validNames.add(tool.name); + return true; + }); + if (body.tool_choice && typeof body.tool_choice === "object" && !Array.isArray(body.tool_choice)) { + if (body.tool_choice.type === "function") { + const n = typeof body.tool_choice.name === "string" ? body.tool_choice.name.trim() : ""; + if (!n || !validNames.has(n)) delete body.tool_choice; + } + } +} + +function sanitizeResponsesItems(body) { + if (!Array.isArray(body.input)) return; + body.input = body.input.filter((item) => { + if (!item || typeof item !== "object" || Array.isArray(item)) return true; + // Strip prior-turn reasoning items: OpenCode Free uses public/pooled credentials + // (`Bearer public`) routing to an upstream OpenAI/Console account pool. + // OpenAI Responses API strictly enforces that reasoning `encrypted_content` + // can only be decrypted by the exact caller/account that issued it; sending it + // across different accounts or rotating proxy relays triggers: + // [invalid_request_error] reasoning `encrypted_content` was not issued to this caller (400). + // Furthermore, under stateless mode (store=false), omitting encrypted_content + // causes OpenAI to reject the referenced reasoning item as "not found or was deleted". + // Dropping prior reasoning items allows multi-turn conversations and tool-calling + // loops to succeed cleanly. + if (item.type === "reasoning") return false; + delete item.encrypted_content; + delete item.reasoning_encrypted_content; + if (item.type === "function_call") { + if (!item.name || typeof item.name !== "string" || item.name.trim() === "") return false; + item.name = item.name.trim().slice(0, MAX_TOOL_NAME_LEN); + item.call_id = clampResponsesCallId(item.call_id); + item.arguments = coerceResponsesArguments(item.arguments); + return true; + } + if (item.type === "function_call_output") { + item.call_id = clampResponsesCallId(item.call_id); + item.output = coerceResponsesOutput(item.output); + return true; + } + return true; }); } @@ -74,12 +463,35 @@ const IP_LIMIT_BODY = /limit|rate|quota|exhausted|capacity|too many|retry/i; export class OpenCodeExecutor extends BaseExecutor { constructor() { super("opencode", PROVIDERS.opencode); - this._currentSessionId = null; + } + + prepareRequestCredentials({ body, credentials, providerSessionId, clientTool } = {}) { + const sourceCredentials = credentials || {}; + const session = resolveOpencodeSession(body, sourceCredentials, providerSessionId, clientTool); + + return { + ...sourceCredentials, + [SESSION_FIELD]: session, + [REQ_FIELD]: resolveOpencodeRequestId(body, sourceCredentials, session), + }; } transformRequest(model, body, stream, credentials) { - this._currentSessionId = resolveOpencodeSession(body, credentials); - if (isResponsesModel(model)) { + if (body && typeof body === "object" && model && !body.model) body.model = model; + // Zen rejects non-streaming requests on free models with 403 FreeTierError; + // always stream upstream and let the handler layer aggregate for non-stream clients. + if (body && typeof body === "object") body.stream = true; + if (isResponsesModel(model || body?.model) && body && typeof body === "object") { + // ponytail: chỉ model đã xác nhận auto-only; mở allowlist khi có bằng chứng. + if ("tool_choice" in body && body.tool_choice !== "auto" + && this.config.quirks?.forceAutoToolChoiceModels?.includes(baseModelId(model))) { + body.tool_choice = "auto"; + } + const normalized = normalizeResponsesInput(body.input); + if (normalized) body.input = normalized; + if (!Array.isArray(body.input) || body.input.length === 0) { + body.input = [{ type: "message", role: "user", content: [{ type: "input_text", text: "..." }] }]; + } // Responses API names the output cap max_output_tokens and takes thinking // as reasoning:{effort,summary} — normalize the Chat fields at this boundary. if (body.max_output_tokens === undefined) { @@ -89,35 +501,54 @@ export class OpenCodeExecutor extends BaseExecutor { delete body.max_tokens; delete body.max_completion_tokens; normalizeOpencodeReasoning(model, body); + body.stream = true; + body.store = false; + normalizeResponsesTools(body); + sanitizeResponsesItems(body); + if (!Array.isArray(body.tools) || body.tools.length === 0) { + cloakOpencodeTools(body, true); + } + } else if (body && typeof body === "object") { + cloakOpencodeTools(body, false); } return injectReasoningContent({ provider: this.provider, model, body }); } - buildUrl(model) { - const base = this.config.baseUrl; - return isResponsesModel(model) - ? `${base}/zen/v1/responses` - : `${base}/zen/v1/chat/completions`; + async execute(args) { + return super.execute({ ...args, credentials: this.prepareRequestCredentials(args) }); } - buildHeaders(credentials, stream = true) { + buildUrl(model) { + const base = this.config.baseUrl; + if (isResponsesModel(model)) return `${base}/zen/v1/responses`; + if (isMessagesModel(model)) return `${base}/zen/v1/messages`; + return `${base}/zen/v1/chat/completions`; + } + + buildHeaders(credentials, stream = true, url = "") { const raw = credentials?.rawHeaders || {}; const lower = {}; for (const [k, v] of Object.entries(raw)) lower[k.toLowerCase()] = v; const downstreamUa = lower["user-agent"] || ""; - const isOpencodeDownstream = downstreamUa.toLowerCase().includes("opencode"); + const isOpencodeDownstream = hasValidOpencodeVersion(downstreamUa); - return { + const session = credentials?.[SESSION_FIELD] || this.prepareRequestCredentials({ credentials })[SESSION_FIELD]; + const downstreamReq = normalizeRequestId(lower["x-opencode-request"]); + const requestId = credentials?.[REQ_FIELD] || downstreamReq || generateRequestId(); + + const headers = { "Content-Type": "application/json", "Authorization": "Bearer public", "User-Agent": isOpencodeDownstream ? downstreamUa : OPENCODE_UA, "x-opencode-client": lower["x-opencode-client"] || "desktop", - "x-opencode-session": lower["x-opencode-session"] || this._currentSessionId || generateSessionId(), - "x-opencode-request": lower["x-opencode-request"] || generateRequestId(), + "x-opencode-session": session, + "x-opencode-request": requestId, "x-opencode-project": lower["x-opencode-project"] || "global", "Accept": stream ? "text/event-stream" : "*/*", }; + if (url.endsWith("/messages")) headers["anthropic-version"] = ANTHROPIC_API_VERSION; + return headers; } parseError(response, bodyText) { diff --git a/open-sse/executors/qoder.js b/open-sse/executors/qoder.js index e52a00df..8ee21ac5 100644 --- a/open-sse/executors/qoder.js +++ b/open-sse/executors/qoder.js @@ -31,16 +31,21 @@ import { proxyAwareFetch } from "../utils/proxyFetch.js"; import { SSE_DONE } from "../utils/sseConstants.js"; import { FETCH_CONNECT_TIMEOUT_MS } from "../config/runtimeConfig.js"; import { - QODER_CHAT_URL_ENCODED, - QODER_CHAT_BASE_ALT, QODER_CHAT_SIG_PATH, - QODER_MODEL_MAP, + QODER_CONTEXT_TIER_ENV, + qoderInferenceBase, } from "../shared/qoder/constants.js"; import { getQoderModelConfig, resolveQoderModels, isQoderPat, resolveQoderCredentials } from "../services/qoderModels.js"; +import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../translator/schema/blocks.js"; +import { encodeDataUri } from "../translator/concerns/image.js"; +import { createQoderSseCoalescer } from "../shared/qoder/sse.js"; +import { rewriteQoderMessageAttachments } from "../shared/qoder/attachments.js"; +import { resolveQoderContextTier, applyQoderContextTier } from "../shared/qoder/contextTier.js"; /** * Hoist role:"system" messages out of the messages array (Qoder rejects - * system in messages) and flatten any multipart content arrays. + * system in messages) and flatten multipart content arrays — EXCEPT image + * blocks, which are preserved (see normalizeContent). */ function normalizeMessages(messages) { if (!Array.isArray(messages) || messages.length === 0) { @@ -50,18 +55,88 @@ function normalizeMessages(messages) { const out = []; for (const msg of messages) { if (!msg || typeof msg !== "object") continue; - const text = extractText(msg.content); if (msg.role === "system") { + const text = extractText(msg.content); if (text) systemParts.push(text); continue; } const cloned = { ...msg }; - cloned.content = text; + cloned.content = normalizeContent(msg.content); out.push(cloned); } return { messages: out, systemText: systemParts.join("\n\n") }; } +/** + * Normalize one message's content for Qoder. + * + * Text-only content is flattened to a plain string (Qoder's historical + * shape). When images are present the content stays an array and image + * blocks are kept as OpenAI-style `image_url` parts. Native qodercli + * uploads inlined bytes to `/api/v2/image/upload` first and then sends + * the OSS URL — `buildQoderRequestBody` does that rewrite before this + * runs. Tiny leftover data URIs are still accepted. The legacy + * top-level `image_urls` / `chat_context.imageUrls` slots stay null — + * qodercli leaves them null too. + * + * Claude-style `{type:"image", source:{...}}` blocks are converted to + * `image_url`. File/document blocks that survived rewrite become short + * stubs so 30MB PDFs never land in agent_chat_generation. + */ +function normalizeContent(content) { + if (typeof content === "string") return content; + if (content == null) return ""; + if (!Array.isArray(content)) return String(content); + + const blocks = []; + const textParts = []; + let hasImage = false; + + const pushText = (text) => { + if (!text) return; + if (hasImage || blocks.length) blocks.push({ type: OPENAI_BLOCK.TEXT, text }); + else textParts.push(text); + }; + + const imageUrlOf = (item) => { + if (typeof item.image_url === "string" && item.image_url) return item.image_url; + if (typeof item.image_url?.url === "string" && item.image_url.url) return item.image_url.url; + return null; + }; + + for (const item of content) { + if (!item || typeof item !== "object") continue; + const imageUrl = item.type === OPENAI_BLOCK.IMAGE_URL ? imageUrlOf(item) : null; + if (imageUrl) { + blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url: imageUrl } }); + hasImage = true; + } else if (item.type === CLAUDE_BLOCK.IMAGE && item.source) { + // Claude base64/url image → OpenAI image_url equivalent. + const src = item.source; + const url = src.type === "base64" && src.data + ? encodeDataUri(src.media_type || "image/png", src.data) + : typeof src.url === "string" && src.url ? src.url : null; + if (url) { + blocks.push({ type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } }); + hasImage = true; + } + } else if (item.type === OPENAI_BLOCK.FILE) { + const name = item.file?.filename || item.file?.name || "file"; + pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`); + } else if (item.type === CLAUDE_BLOCK.DOCUMENT) { + const name = item.title || "document"; + pushText(`[file omitted: ${name} — Qoder reads documents via its file API, not inlined bytes]`); + } else if (typeof item.text === "string" && item.text) { + pushText(item.text); + } + } + + if (!hasImage) return textParts.join("\n"); + // Prepend any text collected before the first image block. + if (textParts.length) blocks.unshift({ type: OPENAI_BLOCK.TEXT, text: textParts.join("\n") }); + return blocks; +} + function extractText(content) { if (typeof content === "string") return content; if (content == null) return ""; @@ -84,9 +159,9 @@ function extractText(content) { function lastUserText(messages) { for (let i = messages.length - 1; i >= 0; i--) { const m = messages[i]; - if (m?.role === "user" && typeof m.content === "string") { - return m.content; - } + if (m?.role !== "user") continue; + if (typeof m.content === "string") return m.content; + if (Array.isArray(m.content)) return extractText(m.content); } return ""; } @@ -110,6 +185,11 @@ function stableChatRecordId(model, messages, tools, maxTokens) { if (m.role) { h.update("\0"); h.update(m.role); } if (typeof m.content === "string" && m.content) { h.update("\0"); h.update(m.content); + } else if (Array.isArray(m.content)) { + // Include image refs so the same prompt with a different image gets + // a distinct chat_record_id. + h.update("\0"); + try { h.update(JSON.stringify(m.content)); } catch {} } } if (tools) { @@ -127,7 +207,7 @@ function truncate(s, n) { /** * Map the OpenAI-style request body into the exact shape Qoder expects. */ -async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal }) { +async function buildQoderRequestBody({ model, body, credentials, log, proxyOptions, signal, uploadFn = null }) { const qoderKey = String(model || "").replace(/^qoder\//, ""); // Fetch model config from dynamic API instead of relying on static QODER_MODEL_MAP. @@ -146,7 +226,30 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio modelConfig = { ...retried, key: qoderKey }; } - const { messages, systemText } = normalizeMessages(body.messages || []); + const incoming = Array.isArray(body.messages) + ? body.messages.map((m) => { + if (!m || typeof m !== "object") return m; + return { + ...m, + content: Array.isArray(m.content) + ? m.content.map((b) => (b && typeof b === "object" ? { ...b } : b)) + : m.content, + }; + }) + : []; + try { + await rewriteQoderMessageAttachments(incoming, { + credentials, + log, + proxyOptions, + signal, + uploadFn, + }); + } catch (err) { + log?.warn?.("QODER", `attachment rewrite failed: ${err.message}`); + } + + const { messages, systemText } = normalizeMessages(incoming); const tools = body.tools; const isReasoning = !!modelConfig.is_reasoning; const maxOutputTokens = Number(modelConfig.max_output_tokens) || 0; @@ -165,7 +268,21 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio const sessionId = stableHash("qoder-session", psd.userId, qoderKey); const recordId = stableChatRecordId(qoderKey, messages, tools, maxTokens); - return { + // Context-window tier (200K/400K/1M): the IDE picks one from model_config.context_config; + // qodercli-style requests default to the smallest. Escalate when the prompt no longer fits. + const tierChoice = resolveQoderContextTier( + modelConfig, + { system: systemText, messages, tools }, + { preference: process.env[QODER_CONTEXT_TIER_ENV] }, + ); + if (tierChoice) { + log?.info?.( + "QODER", + `context tier ${tierChoice.tier.name} (${tierChoice.tier.tokenCount} tokens, ${tierChoice.reason}) for ~${tierChoice.estimatedTokens} prompt tokens`, + ); + } + + const built = { qoderKey, payload: { request_id: uuidv4(), @@ -213,6 +330,8 @@ async function buildQoderRequestBody({ model, body, credentials, log, proxyOptio }, modelConfig, }; + if (tierChoice) applyQoderContextTier(built.payload, tierChoice.tier); + return built; } /** @@ -276,6 +395,11 @@ async function peekFirstQoderFrame(reader, decoder) { * response.text() which hangs until the socket closes — so on terminal * events we cancel the upstream reader and close our stream immediately. * + * Usage: Qoder puts finish_reason on `delta` and sends token counts on a + * later `choices: []` frame. Downstream OpenAI/Claude clients only read + * usage from the finish chunk, so we coalesce those two frames (see + * createQoderSseCoalescer) before forwarding. + * * NEW: Peek first frame to detect billing blocks (code 112/10605/pricingUrl). * If detected, return 403 response so chatCore marks connection unavailable * and triggers combo fallback instead of leaking error text into chat. @@ -302,6 +426,11 @@ async function wrapQoderSSE(response, model) { const upstreamDrained = peek.upstreamDone === true; const encoder = new TextEncoder(); let doneEmitted = false; + const coalescer = createQoderSseCoalescer({ model, encoder, sseDone: SSE_DONE }); + + const syncDone = () => { + if (coalescer.doneEmitted) doneEmitted = true; + }; // Process one already-extracted SSE line (no trailing newline). const processLine = (line, controller) => { @@ -312,15 +441,17 @@ async function wrapQoderSSE(response, model) { const data = trimmed.slice(5).trimStart(); if (data === "[DONE]") { - controller.enqueue(encoder.encode(SSE_DONE)); - doneEmitted = true; + coalescer.flush(controller); + syncDone(); return; } let envelope; try { envelope = JSON.parse(data); } catch { return; } const statusVal = typeof envelope.statusCodeValue === "number" ? envelope.statusCodeValue : 200; - const inner = typeof envelope.body === "string" ? envelope.body : ""; + const inner = typeof envelope.body === "string" + ? envelope.body + : envelope.body != null ? JSON.stringify(envelope.body) : ""; if (statusVal !== 200) { const msg = inner || `upstream status ${statusVal}`; const errChunk = JSON.stringify({ @@ -336,14 +467,8 @@ async function wrapQoderSSE(response, model) { return; } if (!inner) return; - if (inner === "[DONE]") { - controller.enqueue(encoder.encode(SSE_DONE)); - doneEmitted = true; - return; - } - // Strip embedded newlines so the SSE frame stays a single event. - const sanitized = inner.replace(/\r?\n/g, ""); - controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`)); + coalescer.handleInner(inner, controller); + syncDone(); }; const stream = new ReadableStream({ @@ -402,7 +527,7 @@ async function wrapQoderSSE(response, model) { } finally { if (!doneEmitted) { try { - controller.enqueue(encoder.encode(SSE_DONE)); + coalescer.flush(controller); doneEmitted = true; } catch { /* already closed */ } } @@ -431,13 +556,7 @@ export class QoderExecutor extends BaseExecutor { } buildUrl(credentials) { - // Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt- - // with "Login expired" (403). Device tokens (dt-...) stay on api3. - const raw = credentials?.apiKey || credentials?.accessToken; - if (typeof raw === "string" && !raw.startsWith("pt-") && (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-"))) { - return `${QODER_CHAT_BASE_ALT}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`; - } - return QODER_CHAT_URL_ENCODED; + return `${qoderInferenceBase(credentials)}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common&Encode=1`; } // Override execute entirely — Qoder needs: diff --git a/open-sse/executors/xiaomi-mimo.js b/open-sse/executors/xiaomi-mimo.js new file mode 100644 index 00000000..8b69412a --- /dev/null +++ b/open-sse/executors/xiaomi-mimo.js @@ -0,0 +1,99 @@ +import { DefaultExecutor } from "./default.js"; +import { getMimoAccountCookie, invalidateMimoAccountCookieCache, MIMO_API_BASE, MIMO_API_UA } from "../shared/mimoAccount.js"; + +// Desktop-exclusive Preview models. These are served by the account service's +// /api/route proxy, authorized by the Xiaomi account session (NOT the sk- key). +// See shared/mimoAccount.js for the session handshake. +const PREVIEW_MODELS = new Set(["mimo-x-pro-preview", "mimo-x-flash-preview"]); + +// Session cookie resolved in execute() (async) and read back by buildHeaders() +// (sync — BaseExecutor.execute does not await it). Carried on the per-request +// credentials object, same as runtimeTransport. +const COOKIE_KEY = "__mimoAccountCookie"; + +// Upstream calls may hand us either the bare id or a `provider/model` ref. +function bareModel(model) { + const s = String(model || ""); + const i = s.indexOf("/"); + return i >= 0 ? s.slice(i + 1) : s; +} + +export class XiaomiMimoExecutor extends DefaultExecutor { + constructor() { + super("xiaomi-mimo"); + } + + static isPreviewModel(model) { + return PREVIEW_MODELS.has(bareModel(model)); + } + + buildUrl(model, stream, urlIndex = 0, credentials = null) { + // Preview models live on the account-service route, which is not one of the + // declared transports — resolve it before the default runtimeTransport path. + if (XiaomiMimoExecutor.isPreviewModel(model)) { + return `${MIMO_API_BASE}/api/route/chat/completions`; + } + // Cloud API models keep default handling, so a Claude-format client reaches + // the /anthropic/v1/messages transport. + return super.buildUrl(model, stream, urlIndex, credentials); + } + + buildHeaders(credentials, stream = true, url, model) { + if (XiaomiMimoExecutor.isPreviewModel(model) && credentials?.[COOKIE_KEY]) { + // Preview models authenticate with the account-session cookie, not the key. + return { + "Content-Type": "application/json", + Accept: stream ? "text/event-stream" : "application/json", + "User-Agent": MIMO_API_UA, + Cookie: credentials[COOKIE_KEY], + }; + } + return super.buildHeaders(credentials, stream, url, model); + } + + transformRequest(model, body, stream, credentials) { + // super runs stripUnsupportedParams, which flattens Preview content-part + // arrays (see the xiaomi-mimo rule in translator/concerns/paramSupport.js). + const out = super.transformRequest(model, body, stream, credentials); + + // Preview models: thinking/params get defaults only — never override what the + // caller set explicitly. (body.model is already `xiaomi/` via upstreamModelId.) + if (XiaomiMimoExecutor.isPreviewModel(model)) { + if (out.thinking == null) out.thinking = { type: "enabled" }; + if (out.temperature == null) out.temperature = 1.0; + if (out.top_p == null) out.top_p = 0.95; + if (!out.max_tokens) out.max_tokens = 4096; + } + + return out; + } + + async execute(args) { + const { model, credentials, proxyOptions = null } = args; + if (!XiaomiMimoExecutor.isPreviewModel(model)) return super.execute(args); + + const cookie = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions); + if (!cookie) { + throw new Error( + "Xiaomi MiMo account session unavailable. Sign in to MiMo Desktop once so its passToken is present, then retry.", + ); + } + credentials[COOKIE_KEY] = cookie; + const result = await super.execute(args); + + // A cached session can expire early — drop it and retry once with a fresh one. + if (result.response.status === 401) { + invalidateMimoAccountCookieCache(); + const fresh = await getMimoAccountCookie(credentials?.providerSpecificData, proxyOptions).catch(() => null); + if (fresh) { + credentials[COOKIE_KEY] = fresh; + return super.execute(args); + } + } + return result; + } +} + +export const __test__ = { PREVIEW_MODELS, bareModel, COOKIE_KEY }; + +export default XiaomiMimoExecutor; diff --git a/open-sse/executors/zed.js b/open-sse/executors/zed.js index e6233fcb..b88d2762 100644 --- a/open-sse/executors/zed.js +++ b/open-sse/executors/zed.js @@ -29,11 +29,18 @@ import { zedLlmFetch, } from "../shared/zedAuth.js"; +// Wire values for the `provider` field of POST /completions. These are NOT +// display names: cloud.zed.dev matches them exactly, and an unrecognized value +// fails the whole request with `500 {"message":"An internal server error +// occurred."}` before the model is ever looked at. Spellings come from Zed's +// own GET /models catalog: `anthropic`, `open_ai`, `google` (note underscore), +// `x_ai` follows the same convention — so feeding a catalog value back through +// normalizeZedProvider is identity. const ZED_PROVIDER = { - anthropic: "Anthropic", - openai: "OpenAi", - google: "Google", - xai: "XAi", + anthropic: "anthropic", + openai: "open_ai", + google: "google", + xai: "x_ai", }; function normalizeZedProvider(value, model) { @@ -55,7 +62,14 @@ function buildProviderRequest(provider, model, body, stream, credentials) { return openaiToClaudeRequest(model, body, true); } if (provider === ZED_PROVIDER.google) { - return openaiToGeminiRequest(model, body, true); + const geminiRequest = openaiToGeminiRequest(model, body, true); + // Zed's hosted Gemini backend speaks the Vertex safety vocabulary, not the + // public Gemini API enum the shared translator emits (`OFF`, `CIVIC_INTEGRITY`, + // `DANGEROUS_CONTENT`). Drop client-side safetySettings for the Zed Google + // path so Zed applies its own defaults — scoped here so native Gemini/ + // Antigravity is untouched. + delete geminiRequest.safetySettings; + return geminiRequest; } if (provider === ZED_PROVIDER.openai) { return openaiToOpenAIResponsesRequest(model, body, true, credentials); diff --git a/open-sse/handlers/chatCore.js b/open-sse/handlers/chatCore.js index 60c1cee6..49478745 100644 --- a/open-sse/handlers/chatCore.js +++ b/open-sse/handlers/chatCore.js @@ -28,7 +28,7 @@ import { compressWithPxpipe } from "../rtk/pxpipe.js"; import { getCapabilitiesForModel } from "../providers/capabilities.js"; import { stripUnsupportedModalities } from "../translator/concerns/modality.js"; import { prefetchRemoteImages } from "../translator/concerns/prefetch.js"; -import { defaultClaudeToolType } from "../translator/concerns/toolCall.js"; +import { defaultClaudeToolType, shouldDefaultClaudeToolType } from "../translator/concerns/toolCall.js"; import { resolveSessionId } from "../utils/sessionManager.js"; import { markPoolUnfit, clearPoolUnfit } from "../services/proxyPoolFitness.js"; @@ -249,7 +249,11 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred // Claude tool schema requires `type` to be explicitly set; strict gateways (e.g., MiniMax) // reject legacy payloads that omit it with HTTP 400. Default to "custom" when missing. - if (finalFormat === FORMATS.CLAUDE && Array.isArray(translatedBody.tools)) { + // Provider-scoped via quirks (shouldDefaultClaudeToolType): only gateways that declare + // requireClaudeToolType get the explicit type. Applying it unconditionally breaks + // Claude-format endpoints that only accept the legacy typeless tool shape — DeepSeek's + // Anthropic-compatible endpoint 400s with "unknown variant `custom`" (#3905). + if (shouldDefaultClaudeToolType(provider, finalFormat, translatedBody.tools, PROVIDERS)) { translatedBody.tools = defaultClaudeToolType(translatedBody.tools); } @@ -416,7 +420,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred const executeWithPoolFallback = async (attempt = 0) => { let result; try { - result = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions }); + result = await executor.execute({ model, body: translatedBody, stream, credentials, providerSessionId: sessionSeed, clientTool, signal: streamController.signal, log, proxyOptions }); } catch (error) { if (error?.poolScoped && typeof resolveProxyConfig === "function" && attempt < MAX_POOL_RETRIES) { if (await tryNextPool(error.poolScoped, error.message)) return executeWithPoolFallback(attempt + 1); @@ -497,7 +501,17 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred try { await onCredentialsRefreshed(newCredentials); } catch (e) { log?.warn?.("TOKEN", `onCredentialsRefreshed failed: ${e.message}`); } } try { - const retryResult = await executor.execute({ model, body: translatedBody, stream, credentials, signal: streamController.signal, log, proxyOptions }); + const retryResult = await executor.execute({ + model, + body: translatedBody, + stream, + credentials, + providerSessionId: sessionSeed, + clientTool, + signal: streamController.signal, + log, + proxyOptions, + }); if (retryResult.response.ok) { providerResponse = retryResult.response; providerUrl = retryResult.url; @@ -556,7 +570,7 @@ export async function handleChatCore({ body, modelInfo, credentials, log, onCred // Streaming response const { onStreamComplete, streamDetailId } = buildOnStreamComplete({ ...sharedCtx }); - return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId }); + return handleStreamingResponse({ ...sharedCtx, providerResponse, sourceFormat, targetFormat: providerResponseFormat, userAgent, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, credentials }); } export function isTokenExpiringSoon(expiresAt, bufferMs = 5 * 60 * 1000) { diff --git a/open-sse/handlers/chatCore/nonStreamingHandler.js b/open-sse/handlers/chatCore/nonStreamingHandler.js index 1b4e262c..dab253c7 100644 --- a/open-sse/handlers/chatCore/nonStreamingHandler.js +++ b/open-sse/handlers/chatCore/nonStreamingHandler.js @@ -6,6 +6,7 @@ import { addBufferToUsage, filterUsageForFormat } from "../../utils/usageTrackin import { createErrorResult } from "../../utils/error.js"; import { HTTP_STATUS } from "../../config/runtimeConfig.js"; import { parseSSEToOpenAIResponse } from "./sseToJsonHandler.js"; +import { unwrapClineEnvelope } from "../../shared/clineEnvelope.js"; import { buildRequestDetail, extractRequestConfig, extractUsageFromResponse, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { appendRequestLog, saveRequestDetail } from "@/lib/usageDb.js"; import { decloakToolNames } from "../../utils/claudeCloaking.js"; @@ -319,6 +320,11 @@ export async function handleNonStreamingResponse({ providerResponse, provider, m } } + // Unwrap before any consumer reads choices/usage so non-stream clients get a + // bare OpenAI body and usage tracking sees data.usage. No-op unless the + // provider opts in via transport.quirks.clineEnvelope. + responseBody = unwrapClineEnvelope(responseBody, provider); + reqLogger.logProviderResponse(providerResponse.status, providerResponse.statusText, providerResponse.headers, responseBody); // Unwrap AFTER logging (raw envelope stays in the log for forensics) but // BEFORE usage extraction/translation so choices/usage resolve downstream. diff --git a/open-sse/handlers/chatCore/streamingHandler.js b/open-sse/handlers/chatCore/streamingHandler.js index be008e7d..6751256b 100644 --- a/open-sse/handlers/chatCore/streamingHandler.js +++ b/open-sse/handlers/chatCore/streamingHandler.js @@ -3,8 +3,9 @@ import { needsTranslation } from "../../translator/index.js"; import { createSSETransformStreamWithLogger, createPassthroughStreamWithLogger } from "../../utils/stream.js"; import { pipeWithDisconnect } from "../../utils/streamHandler.js"; import { PROVIDERS } from "../../config/providers.js"; -import { STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js"; +import { HTTP_STATUS, STREAM_STALL_TIMEOUT_MS } from "../../config/runtimeConfig.js"; import { buildAbortedResponsesTerminalBytes } from "../../utils/responsesStreamHelpers.js"; +import { buildStreamErrorBytes } from "../../utils/streamHelpers.js"; import { buildRequestDetail, extractRequestConfig, saveUsageStats, formatDoneLine } from "./requestDetail.js"; import { saveRequestDetail } from "@/lib/usageDb.js"; import { SSE_HEADERS_CORS as SSE_HEADERS } from "../../utils/sseConstants.js"; @@ -22,7 +23,7 @@ const CODEX_SOURCE_TO_TARGET = { /** * Determine which SSE transform stream to use based on provider/format. */ -function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }) { +function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }) { const isDroidCLI = userAgent?.toLowerCase().includes("droid") || userAgent?.toLowerCase().includes("codex-cli"); // Responses-API providers (e.g. codex) emit Responses SSE → translate into client format const isResponsesProvider = PROVIDERS[provider]?.format === FORMATS.OPENAI_RESPONSES; @@ -30,11 +31,11 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, if (needsCodexTranslation) { const codexTarget = CODEX_SOURCE_TO_TARGET[sourceFormat] || FORMATS.OPENAI; - return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames); + return createSSETransformStreamWithLogger(FORMATS.OPENAI_RESPONSES, codexTarget, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials); } if (needsTranslation(targetFormat, sourceFormat)) { - return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames); + return createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider, reqLogger, toolNameMap, model, connectionId, body, onStreamComplete, apiKey, customToolNames, credentials); } return createPassthroughStreamWithLogger(provider, reqLogger, model, connectionId, body, onStreamComplete, apiKey); @@ -43,7 +44,7 @@ function buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, /** * Handle streaming response — pipe provider SSE through transform stream to client. */ -export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log }) { +export async function handleStreamingResponse({ providerResponse, provider, model, sourceFormat, targetFormat, userAgent, body, stream, translatedBody, finalBody, requestStartTime, connectionId, apiKey, clientRawRequest, onRequestSuccess, reqLogger, toolNameMap, customToolNames, streamController, onStreamComplete, streamDetailId, pxpipe, reqTag, log, credentials }) { if (onRequestSuccess) { Promise.resolve() .then(onRequestSuccess) @@ -79,11 +80,16 @@ export async function handleStreamingResponse({ providerResponse, provider, mode }; } - const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey }); + const transformStream = buildTransformStream({ provider, sourceFormat, targetFormat, userAgent, reqLogger, toolNameMap, customToolNames, model, connectionId, body, onStreamComplete, apiKey, credentials }); - // Responses passthrough: synthesize response.failed + [DONE] if the stream aborts/stalls before a terminal event + // Terminal bytes when the stream aborts after HTTP 200 was already sent, so the + // client sees a real error instead of a silently truncated stream. + // Responses passthrough keeps its own response.failed shape; every other client + // format gets the OpenAI error frame + [DONE], or `event: error` for Claude. const isResponsesPassthrough = sourceFormat === FORMATS.OPENAI_RESPONSES && targetFormat === FORMATS.OPENAI_RESPONSES; - const onAbortTerminal = isResponsesPassthrough ? buildAbortedResponsesTerminalBytes : null; + const onAbortTerminal = isResponsesPassthrough + ? buildAbortedResponsesTerminalBytes + : (message) => buildStreamErrorBytes(HTTP_STATUS.GATEWAY_TIMEOUT, message, sourceFormat); const stallTimeoutMs = PROVIDERS[provider]?.stallTimeoutMs || STREAM_STALL_TIMEOUT_MS; const transformedBody = pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal, stallTimeoutMs); diff --git a/open-sse/handlers/imageProviders/codex.js b/open-sse/handlers/imageProviders/codex.js index 218302ab..afaf8e97 100644 --- a/open-sse/handlers/imageProviders/codex.js +++ b/open-sse/handlers/imageProviders/codex.js @@ -2,13 +2,21 @@ import { randomUUID } from "node:crypto"; import { nowSec } from "./_base.js"; import { PROVIDERS } from "../../config/providers.js"; +import { CODEX_CLI_VERSION } from "../../config/appConstants.js"; const CODEX_RESPONSES_URL = PROVIDERS["codex"].baseUrl; -const CODEX_USER_AGENT = "codex_cli_rs/0.136.0"; -const CODEX_VERSION = "0.136.0"; +const CODEX_USER_AGENT = `codex_cli_rs/${CODEX_CLI_VERSION}`; const CODEX_ORIGINATOR = "codex_cli_rs"; const CODEX_MODEL_SUFFIX = "-image"; const CODEX_REF_DETAIL = "high"; +const CODEX_IMAGES_MAIN_MODEL = "gpt-5.5"; +const CODEX_TOOL_IMAGE_MODELS = new Set([ + "gpt-image-1.5", + "gpt-image-2", + "gpt-image-2.5", + "gpt-image-2.5-flare", + "gpt-image-2.5-sunburst", +]); function decodeAccountId(idToken) { try { @@ -27,6 +35,13 @@ function stripImageSuffix(model) { return model.endsWith(CODEX_MODEL_SUFFIX) ? model.slice(0, -CODEX_MODEL_SUFFIX.length) : model; } +function resolveCodexImageModels(model) { + if (CODEX_TOOL_IMAGE_MODELS.has(model)) { + return { responsesModel: CODEX_IMAGES_MAIN_MODEL, toolModel: model }; + } + return { responsesModel: stripImageSuffix(model), toolModel: null }; +} + function toDataUrl(input) { if (!input || typeof input !== "string") return null; if (/^data:image\//i.test(input) || /^https?:\/\//i.test(input)) return input; @@ -157,7 +172,7 @@ export default { "originator": CODEX_ORIGINATOR, "session_id": randomUUID(), "user-agent": CODEX_USER_AGENT, - "version": CODEX_VERSION, + "version": CODEX_CLI_VERSION, "x-client-request-id": randomUUID(), }; }, @@ -167,21 +182,26 @@ export default { const single = toDataUrl(body.image); if (single) refs.push(single); const detail = body.image_detail || CODEX_REF_DETAIL; + const { responsesModel, toolModel } = resolveCodexImageModels(model); const imgTool = { type: "image_generation", output_format: (body.output_format || "png").toLowerCase() }; + if (toolModel) { + imgTool.action = refs.length > 0 ? "edit" : "generate"; + imgTool.model = toolModel; + } if (body.size && body.size !== "") imgTool.size = body.size; if (body.quality && body.quality !== "") imgTool.quality = body.quality; if (body.background && body.background !== "") imgTool.background = body.background; return { - model: stripImageSuffix(model), + model: responsesModel, instructions: "", input: [{ type: "message", role: "user", content: buildContent(body.prompt, refs, detail) }], tools: [imgTool], - tool_choice: "auto", + tool_choice: toolModel ? { type: "image_generation" } : "auto", parallel_tool_calls: false, prompt_cache_key: randomUUID(), stream: true, store: false, - reasoning: null, + reasoning: toolModel ? { effort: "medium", summary: "auto" } : null, }; }, // Custom: codex parses SSE → either pipe to client or collect b64 diff --git a/open-sse/handlers/videoCore.js b/open-sse/handlers/videoCore.js index 98d60157..45f14d86 100644 --- a/open-sse/handlers/videoCore.js +++ b/open-sse/handlers/videoCore.js @@ -2,6 +2,7 @@ import { createErrorResult } from "../utils/error.js"; import { HTTP_STATUS } from "../config/runtimeConfig.js"; import { refreshTokenByProvider } from "../services/tokenRefresh.js"; import { PROVIDER_MEDIA } from "../providers/index.js"; +import { getVideoAdapter } from "./videoProviders/index.js"; // Upstream fetch deadline for video job submission/polling (the job itself is // async upstream — this only bounds the HTTP round-trip, not video rendering). @@ -94,21 +95,49 @@ export async function handleVideoProxyCore({ return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Unknown video action: ${action}`); } - const method = requestId ? "GET" : "POST"; - const url = buildUpstreamUrl(config, action, requestId); + const adapter = getVideoAdapter(provider); const fetchSignal = combineSignals(signal, timeoutMs); - const doFetch = (token) => - fetch(url, { + // Default (xAI shape) request plan; adapters override URL/method/headers/body. + const defaultPlan = () => { + const method = requestId ? "GET" : "POST"; + return { method, - headers: buildHeaders({ token, contentType: method === "POST" ? contentType : null, idempotencyKey: method === "POST" ? idempotencyKey : null }), + url: buildUpstreamUrl(config, action, requestId), + headers: buildHeaders({ + token: credentials?.accessToken || credentials?.apiKey, + contentType: method === "POST" ? contentType : null, + idempotencyKey: method === "POST" ? idempotencyKey : null, + }), body: method === "POST" ? rawBody : undefined, - signal: fetchSignal, - }); + }; + }; + // Rebuilt per attempt so the auth retry below picks up the refreshed token. + const doFetch = async () => { + const plan = adapter + ? await adapter.buildRequest({ + config, action, requestId, rawBody, contentType, idempotencyKey, credentials, log, + token: credentials?.accessToken || credentials?.apiKey, + }) + : defaultPlan(); + if (plan.error) return { planError: plan.error }; + return { + response: await fetch(plan.url, { + method: plan.method, + headers: plan.headers, + body: plan.body, + signal: fetchSignal, + }), + }; + }; + + const method = requestId ? "GET" : "POST"; let upstream; try { - upstream = await doFetch(credentials?.accessToken || credentials?.apiKey); + const first = await doFetch(); + if (first.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${first.planError}`); + upstream = first.response; } catch (error) { if (error?.name === "AbortError" || error?.name === "TimeoutError") { return createErrorResult(HTTP_STATUS.REQUEST_TIMEOUT, `[${provider}] video ${method} aborted: ${error.message}`); @@ -136,7 +165,9 @@ export async function handleVideoProxyCore({ await upstream.body?.cancel?.(); } catch { /* noop */ } try { - upstream = await doFetch(credentials.accessToken || credentials.apiKey); + const retry = await doFetch(); + if (retry.planError) return createErrorResult(HTTP_STATUS.BAD_REQUEST, `[${provider}] ${retry.planError}`); + upstream = retry.response; } catch (error) { return createErrorResult(HTTP_STATUS.BAD_GATEWAY, sanitizeSecrets(`[${provider}] video retry after refresh failed: ${error.message}`, credentials)); } @@ -152,13 +183,25 @@ export async function handleVideoProxyCore({ return createErrorResult(upstream.status, `[${provider}] ${message.slice(0, 2000)}`); } - // Success: pass the upstream JSON through untouched (request_id / status / video.url). + // Success: pass the upstream JSON through untouched (request_id / status / video.url), + // unless the adapter maps a provider-native shape onto it (Vertex operations). + let outBody = bodyText; + let outType = upstream.headers.get("content-type") || "application/json"; + if (adapter?.transformResponse) { + try { + outBody = JSON.stringify(adapter.transformResponse(JSON.parse(bodyText))); + outType = "application/json"; + } catch { + // Non-JSON or unexpected shape — fall back to the raw upstream body. + } + } + return { success: true, - response: new Response(bodyText, { + response: new Response(outBody, { status: upstream.status, headers: { - "Content-Type": upstream.headers.get("content-type") || "application/json", + "Content-Type": outType, "Access-Control-Allow-Origin": "*", }, }), diff --git a/open-sse/handlers/videoProviders/index.js b/open-sse/handlers/videoProviders/index.js new file mode 100644 index 00000000..28173972 --- /dev/null +++ b/open-sse/handlers/videoProviders/index.js @@ -0,0 +1,13 @@ +// Video provider adapters. +// +// Default (no adapter) = xAI shape: raw body forwarded to {baseUrl}/{action}, +// polled at {baseUrl}/{id}, upstream JSON passed through verbatim. +// A provider only needs an adapter when its wire format differs from that. +import openrouter from "./openrouter.js"; +import vertex from "./vertex.js"; + +const ADAPTERS = { openrouter, vertex }; + +export function getVideoAdapter(provider) { + return ADAPTERS[provider] || null; +} diff --git a/open-sse/handlers/videoProviders/openrouter.js b/open-sse/handlers/videoProviders/openrouter.js new file mode 100644 index 00000000..90198a39 --- /dev/null +++ b/open-sse/handlers/videoProviders/openrouter.js @@ -0,0 +1,39 @@ +// OpenRouter video jobs — https://openrouter.ai/docs/api/api-reference/videos +// +// Same async shape as xAI (POST → { id, status }, GET → status/unsigned_urls), +// two differences only: creation POSTs to the collection root (no `/generations` +// suffix) and the account headers come from the registry entry. +// Response bodies are passed through verbatim. + +// ponytail: generations only — OpenRouter has no edits/extensions endpoint today. +const SUPPORTED_ACTIONS = new Set(["generations"]); + +function headers(config, token) { + return { + Accept: "application/json", + ...(config.headers || {}), + ...(token ? { Authorization: `Bearer ${token}` } : {}), + }; +} + +export default { + buildRequest({ config, action, requestId, rawBody, contentType, token }) { + const base = config.baseUrl.replace(/\/$/, ""); + + if (requestId) { + return { method: "GET", url: `${base}/${encodeURIComponent(requestId)}`, headers: headers(config, token) }; + } + if (!SUPPORTED_ACTIONS.has(action)) { + return { error: `OpenRouter video supports 'generations' only (got '${action}')` }; + } + if (contentType && !contentType.includes("application/json")) { + return { error: "OpenRouter video requires an application/json body" }; + } + return { + method: "POST", + url: base, + headers: { ...headers(config, token), "Content-Type": "application/json" }, + body: rawBody, + }; + }, +}; diff --git a/open-sse/handlers/videoProviders/vertex.js b/open-sse/handlers/videoProviders/vertex.js new file mode 100644 index 00000000..a4ff66c6 --- /dev/null +++ b/open-sse/handlers/videoProviders/vertex.js @@ -0,0 +1,159 @@ +// Vertex AI (Veo) video jobs. +// +// Vertex does NOT speak the OpenAI-ish /v1/videos shape, so unlike OpenRouter +// this adapter translates both directions: +// create → POST {model}:predictLongRunning { instances[], parameters{} } → { name } +// poll → POST {model}:fetchPredictOperation { operationName } → { done, response } +// Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation +// +// The operation name is a resource path (contains "/"), so it is base64url-encoded +// into the job id returned to the client — GET /v1/videos/{id} stays a flat path. +import { parseVertexSaJson, refreshVertexToken } from "../../services/tokenRefresh.js"; + +const DEFAULT_LOCATION = "us-central1"; + +const encodeJobId = (name) => Buffer.from(name, "utf8").toString("base64url"); + +// Operation name shape: projects/{p}/locations/{l}/publishers/{pub}/models/{m}/operations/{op}. +// Anchored and single-segment-per-field so a decoded path can never carry `..` or a +// host-changing prefix into the request URL. +const OPERATION_NAME_RE = /^projects\/[^/]+\/locations\/[^/]+\/publishers\/[^/]+\/models\/[^/]+\/operations\/[^/]+$/; + +function modelPathOf(operationName) { + return operationName.slice(0, operationName.indexOf("/operations/")); +} + +function decodeJobId(id) { + const raw = String(id ?? ""); + // Buffer.from(x, "base64url") silently drops invalid characters instead of + // throwing, so only ids that re-encode byte-for-byte are accepted. + if (!raw || raw.length > 1024 || !/^[A-Za-z0-9_-]+$/.test(raw)) return null; + const decoded = Buffer.from(raw, "base64url").toString("utf8"); + if (Buffer.from(decoded, "utf8").toString("base64url") !== raw) return null; + return OPERATION_NAME_RE.test(decoded) ? decoded : null; +} + +async function resolveAuth(credentials, log) { + const saJson = parseVertexSaJson(credentials?.apiKey); + const projectId = + saJson?.project_id || + credentials?.projectId || + credentials?.providerSpecificData?.projectId; + const location = credentials?.providerSpecificData?.location || DEFAULT_LOCATION; + + if (!projectId) { + return { error: "Vertex video requires a project_id — use Service Account JSON or set providerSpecificData.projectId" }; + } + + let token = credentials?.accessToken; + if (saJson) { + const minted = await refreshVertexToken(saJson, log); + if (!minted?.accessToken) return { error: "Vertex video: failed to mint access token from service account JSON" }; + token = minted.accessToken; + } + if (!token) return { error: "Vertex video requires Service Account JSON or an OAuth access token (raw API keys are not supported)" }; + + return { token, projectId, location }; +} + +/** OpenAI-ish video body → Vertex predictLongRunning body. */ +function toVertexBody(body) { + const instance = { prompt: body.prompt }; + // Image-to-video: accept the Vertex-native shape or a bare data URL / base64 string. + const image = body.image ?? body.image_url; + if (image && typeof image === "object") { + instance.image = image; + } else if (typeof image === "string") { + const match = image.match(/^data:([^;]+);base64,(.*)$/s); + instance.image = match + ? { bytesBase64Encoded: match[2], mimeType: match[1] } + : { gcsUri: image }; + } + if (body.video && typeof body.video === "object") instance.video = body.video; + + const parameters = {}; + if (body.n != null) parameters.sampleCount = Number(body.n); + if (body.duration != null) parameters.durationSeconds = Number(body.duration); + if (body.aspect_ratio) parameters.aspectRatio = body.aspect_ratio; + if (body.resolution) parameters.resolution = body.resolution; + if (body.seed != null) parameters.seed = body.seed; + if (body.negative_prompt) parameters.negativePrompt = body.negative_prompt; + // Without storageUri Vertex returns inline base64 bytes; a GCS bucket keeps + // the poll response small and is what production callers want. + if (body.storage_uri) parameters.storageUri = body.storage_uri; + if (body.generate_audio != null) parameters.generateAudio = !!body.generate_audio; + + return { instances: [instance], ...(Object.keys(parameters).length ? { parameters } : {}) }; +} + +/** Vertex operation → the async-job shape 9Router clients already poll for. */ +function fromVertexOperation(json) { + if (!json?.name) return json; + const id = encodeJobId(json.name); + if (json.error) { + return { id, request_id: id, status: "failed", error: json.error }; + } + if (!json.done) { + return { id, request_id: id, status: "pending" }; + } + const samples = + json.response?.videos || + json.response?.generateVideoResponse?.generatedSamples || + []; + const videos = samples.map((s) => ({ + url: s.gcsUri || s.video?.uri || s.uri || null, + b64_json: s.bytesBase64Encoded || s.video?.bytesBase64Encoded || null, + mime_type: s.mimeType || s.video?.mimeType || "video/mp4", + })); + return { id, request_id: id, status: "completed", video: videos[0] || null, videos }; +} + +export default { + async buildRequest({ config, action, requestId, rawBody, contentType, credentials, log }) { + if (contentType && !contentType.includes("application/json")) { + return { error: "Vertex video requires an application/json body" }; + } + + const auth = await resolveAuth(credentials, log); + if (auth.error) return { error: auth.error }; + const { token, projectId, location } = auth; + const base = (config.baseUrl || "https://aiplatform.googleapis.com").replace(/\/$/, ""); + const headers = { Accept: "application/json", "Content-Type": "application/json", Authorization: `Bearer ${token}` }; + + if (requestId) { + const operationName = decodeJobId(requestId); + if (!operationName) return { error: "Invalid Vertex video job id" }; + return { + method: "POST", + url: `${base}/v1/${modelPathOf(operationName)}:fetchPredictOperation`, + headers, + body: JSON.stringify({ operationName }), + }; + } + + if (action !== "generations") { + // ponytail: Veo extend/edit go through generations with `video`/`image` in the body. + return { error: `Vertex video supports 'generations' only (got '${action}')` }; + } + + let body; + try { + body = JSON.parse(typeof rawBody === "string" ? rawBody : rawBody.toString("utf8")); + } catch { + return { error: "Invalid JSON body" }; + } + if (!body.model) return { error: "Vertex video requires a model (e.g. vertex/veo-3.1-generate-preview)" }; + // Plain model id only — a path segment carrying "/" or ".." would rewrite the URL. + if (!/^[A-Za-z0-9._-]+$/.test(body.model)) return { error: "Invalid Vertex video model id" }; + if (!body.prompt && !body.image && !body.image_url) return { error: "Vertex video requires a prompt or an image" }; + + return { + method: "POST", + url: `${base}/v1/projects/${projectId}/locations/${location}/publishers/google/models/${body.model}:predictLongRunning`, + headers, + body: JSON.stringify(toVertexBody(body)), + }; + }, + + transformResponse: fromVertexOperation, +}; diff --git a/open-sse/providers/capabilities.js b/open-sse/providers/capabilities.js index a69d9bbf..1042789b 100644 --- a/open-sse/providers/capabilities.js +++ b/open-sse/providers/capabilities.js @@ -116,6 +116,16 @@ export const MODEL_CAPABILITIES = { // DeepSeek's first V4 model with image input; text limits match V4-Flash. "deepseek-v4-flash-vision-exp": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, + // DeepSeek V4.1-Flash is natively multimodal — models.dev lists + // opencode-go/deepseek-v4.1-flash with modalities.input ["text","image"] — and upstream + // the retired v4-flash / vision-exp ids route to it, so the live V4.1 ids carry the + // same image capability as the exp id above. "deepseek-flash" is the GA id on the + // DeepSeek API; it previously fell through to the generic *deepseek* pattern, whose + // 128K/64K limits are kept here. The repeated fields are deliberate: an exact entry + // short-circuits the pattern table, so a vision-only delta would drop them. + "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, + "deepseek-flash": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 128000, maxOutput: 64000 }, + // Qwen plain coder/text (no vision) — registry "vision-model" / "coder-model" aliases "vision-model": { vision: true, reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 }, "coder-model": { reasoning: true, thinkingFormat: "qwen", contextWindow: 1000000 }, @@ -131,6 +141,8 @@ export const MODEL_CAPABILITIES = { // via OpenAI Responses input_image; reasoning supports up to xhigh. "muse-spark-1.2-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 }, "muse-spark-1.3-contributor-free": { vision: true, reasoning: true, thinkingFormat: "openai", contextWindow: 1048576, maxOutput: 131072 }, + // OpenCode Free Union Alpha — multimodal (text+vision), 262K context, 131K max output + "union-alpha": { vision: true, contextWindow: 262144, maxOutput: 131072 }, }; const KIRO_GPT_5_6_CAPABILITIES = { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }; @@ -154,6 +166,7 @@ export const PROVIDER_CAPABILITIES = { "deepseek-ai/deepseek-v4-flash": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 65536 }, }, "codex": { + "gpt-6-astra": { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 }, "gpt-5.6-sol": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-sol-review": CODEX_GPT_56_SOL_CAPS, "gpt-5.6-terra": CODEX_GPT_56_DEFAULT_CAPS, @@ -178,39 +191,95 @@ export const PROVIDER_CAPABILITIES = { // CodeBuddy.cn — authoritative per-model metadata from the gateway's model // config (contextWindow=maxInputTokens, maxOutput=maxOutputTokens, vision= // supportsImages). Every model reasons via OpenAI-style reasoning_effort - // (see registry thinkingFormat). `onlyReasoning` models can't turn thinking - // off → thinkingCanDisable:false (clamped to minimal instead of disabled). + // (see registry thinkingFormat). For thinkingCanDisable use the server's + // reasoning.canDisableThinking flag — see the note in the codebuddy-cn block + // below; it is NOT the inverse of onlyReasoning. "codebuddy-cn": { - "glm-5.2": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 }, - "glm-5.1": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, + "glm-5.2": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 }, + "glm-5.1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, "glm-5.0": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 }, - "glm-5.0-turbo": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, - "glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 38000 }, + // maxOutput 64000 per both the plugin-baked fallback and the live server + // table (the old 38000 had no source and truncated output). + "glm-5v-turbo": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 64000 }, "glm-4.7": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 48000 }, - "minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 48000 }, - "minimax-m2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, + "minimax-m3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 512000, maxOutput: 128000 }, "kimi-k2.7": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, "kimi-k2.6": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, - "kimi-k2.5": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 164000, maxOutput: 32000 }, - "hy3-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 }, - // hy3/hy3-x: 256K official (192K conservative, matches hy3-preview); hy4-preview: 1M official. - // glm-5.3: 1M (GLM-5.x gen); glm-5.3-flash window unverified (200K conservative). + // Per-model values mirror the server's product-config payload (the plugin + // fetches it from copilot.tencent.com; the `models[]` entries carry + // maxInputTokens/maxOutputTokens/supportsImages). contextWindow = + // maxInputTokens, maxOutput = maxOutputTokens. Where the server and the + // plugin-baked fallback disagree, the server table wins. + // ⚠️ thinkingCanDisable maps to the server's reasoning.canDisableThinking — + // it is NOT the inverse of onlyReasoning. onlyReasoning means "thinking is + // on by default"; canDisableThinking means "it CAN be turned off". glm-5.3 + // and glm-5.3-flash are onlyReasoning:true BUT canDisableThinking:true, so + // their thinking is switchable; the hy* models are forced always-on. "hy3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 }, - "hy3-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 192000, maxOutput: 64000 }, "hy4-preview": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 }, - "hy4-preview-x": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 64000 }, - "glm-5.3": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 48000 }, - "glm-5.3-flash": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 200000, maxOutput: 48000 }, - "kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 32000 }, - "deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 }, - "deepseek-v4-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 50000 }, - "deepseek-v3-2-volc": { reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 96000, maxOutput: 32000 }, + "glm-5.3": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 48000 }, + "glm-5.3-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 32000 }, + "kimi-k3-1": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 32000 }, + "deepseek-v4-pro": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 50000 }, + // deepseek-v4.1-flash replaces v4-flash (dropped from the server list; + // the old endpoint still answers 200 but the published list is the + // contract). maxOutput 128000 per the server's product-config payload. + "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 }, + }, + // CodeBuddy intl — same gateway catalog as CN, so deepseek-v4.1-flash mirrors + // the codebuddy-cn entry (the openai-style reasoning_effort format matters: + // the generic *deepseek-v4* pattern would otherwise pick the vendor-native + // "deepseek" thinking shape, which the CodeBuddy gateway does not accept). + "codebuddy-intl": { + "deepseek-v4.1-flash": { vision: true, reasoning: true, thinkingFormat: "openai", thinkingCanDisable: true, contextWindow: 1000000, maxOutput: 128000 }, + }, + // Qoder — upstream exposes opaque internal ids (dfmodel, kmodel, …); the + // registry `name` is display-only and capability lookup matches on the raw + // id, so every qoder model would fall through to DEFAULT_CAPABILITIES + // (200K) without this map. contextWindow follows the real model family's + // spec: the /algo/api/v2/model/list max_input_tokens under-reports some + // windows (GLM-5.3 / Kimi-K3 / Qwen3.8-Max claim 180K but accept more). + // max_output_tokens arrives as 0 for every model, so outputs are + // best-guess from the real model family. Vision tags below follow the + // upstream is_vl flag. The executor uploads inlined images to + // /api/v2/image/upload and leaves image_urls/chat_context.imageUrls null + // (same as qodercli). reasoning:true on all of them — every model can + // reason; the upstream is_reasoning flag only drives model_config selection. + // thinkingFormat keeps the true-model family for documentation/UI, but + // thinkingCanDisable:false everywhere: the executor only forwards + // messages/tools/max_tokens, and thinking is fixed upstream via + // modelConfig.is_reasoning — client thinking intent is dropped, so "none" + // must never be offered as an option. + "qoder": { + "ultimate": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Opus 5 + "performance": { vision: true, reasoning: true, thinkingFormat: "claude-adaptive", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // Claude Sonnet 5 + "dmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Pro + "dfmodel": { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // DeepSeek-V4-Flash + "gmodel": { reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3 + "gfmodel": { vision: true, reasoning: true, thinkingFormat: "zai", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 128000 }, // GLM-5.3-Flash + "kmodel_latest": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Kimi-K3 + "kmodel": { vision: true, reasoning: true, thinkingFormat: "kimi", thinkingCanDisable: false, contextWindow: 256000, maxOutput: 65536 }, // Kimi-K2.7-Code + "mmodel": { reasoning: true, thinkingFormat: "minimax", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 512000 }, // MiniMax-M3 + "qmodel_latest": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Max + "qmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.7-Plus + "qfmodel": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Flash + "qmodel_38max": { vision: true, reasoning: true, thinkingFormat: "qwen", thinkingCanDisable: false, contextWindow: 1000000, maxOutput: 65536 }, // Qwen3.8-Max }, // Poolside Laguna — OpenAI-compatible, all reasoning-capable (32K max output). "poolside": { "laguna-s-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 1000000, maxOutput: 32000 }, "laguna-xs-2.1": { reasoning: true, thinkingFormat: "openai", contextWindow: 200000, maxOutput: 32000 }, }, + // Ollama Cloud — the generic *deepseek-v4* pattern misses the vision badge + // the library page publishes for this model (text+image in, 1M context). + // ponytail: thinkingFormat stays "deepseek" to preserve today's body shape; + // Ollama's native toggle is the top-level `think` field (bool or + // low/medium/high/max), which no format in thinkingUnified.js emits yet — + // openai-to-ollama.js drops it. Wire a "think" format when thinking on + // Ollama Cloud is actually needed. + "ollama": { + "deepseek-v4.1-flash:cloud": { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }, + }, }; /** @@ -247,6 +316,9 @@ export const PATTERN_CAPABILITIES = [ { pattern: "*gemma*", caps: { vision: true, contextWindow: 128000 } }, { pattern: "*nanobanana*", caps: { vision: true, imageOutput: true } }, + // ── OpenAI GPT-6.x (vision + thinking + web search) ────────────── + { pattern: "*gpt-6*", caps: { vision: true, reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 272000, maxOutput: 128000 } }, + // ── OpenAI GPT-5.x (vision + thinking + web search) ────────────── { pattern: "*gpt-5*image*", caps: { imageOutput: true } }, { pattern: "*gpt-5*codex*", caps: { reasoning: true, search: true, thinkingFormat: "openai", contextWindow: 400000, maxOutput: 128000 } }, @@ -312,7 +384,11 @@ export const PATTERN_CAPABILITIES = [ { pattern: "*glm*", caps: { reasoning: true, thinkingFormat: "zai", contextWindow: 200000 } }, // ── DeepSeek (thinking.enabled + reasoning_effort; r1 = thinking-only) ─ - { pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 } }, + // v4.1+ has real image input (probed live on Alibaba MaaS: correct color + // read from a PNG). v4-pro / v4-flash-0731 accept image blocks but ignore + // them (answered "Unknown"), so vision stays scoped to v4.* dotted releases. + { pattern: "*deepseek-v4.*", caps: { vision: true, reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 128000 } }, + { pattern: "*deepseek-v4*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingEffortSupported: true, contextWindow: 1000000, maxOutput: 384000 } }, { pattern: "*reasoner*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } }, { pattern: "*deepseek-r*", caps: { reasoning: true, thinkingFormat: "deepseek", thinkingCanDisable: false, contextWindow: 128000 } }, { pattern: "*deepseek-chat*", caps: { contextWindow: 128000 } }, @@ -377,14 +453,27 @@ const MODALITY_KEYS = ["vision", "pdf", "audioInput", "videoInput"]; // Catalog lookups, installed by the server at startup. Left as no-ops in the // browser bundle, where there is no file to read. +// +// The server bundles this module into every route chunk that needs it, and each +// copy carries its own module state, so an install landing in the copy the +// startup hook imported stays invisible to the copy resolving requests. The slot +// lives on globalThis instead; the local binding is the fast path. let catalogSource = null; /** * Install the synced catalog reader (server only). - * @param {{ getModalities: Function, getLimits: Function } | null} source + * @param {{ getModalities: (provider: string, model: string) => object|null, + * getLimits: (provider: string, model: string) => object|null } | null} source */ export function setCatalogSource(source) { catalogSource = source; + if (typeof globalThis !== "undefined") globalThis.__9rCatalogSource = source; +} + +function getCatalogSource() { + if (catalogSource) return catalogSource; + if (typeof globalThis === "undefined") return null; + return (catalogSource = globalThis.__9rCatalogSource || null); } // Apply the synced catalog + name heuristic on top of a table-resolved result. @@ -393,15 +482,16 @@ export function setCatalogSource(source) { function refine(base, provider, model) { const result = { ...DEFAULT_CAPABILITIES, ...base }; - if (catalogSource) { - const modalities = catalogSource.getModalities(model); + const source = getCatalogSource(); + if (source) { + const modalities = source.getModalities(provider, model); if (modalities) { for (const key of MODALITY_KEYS) { if (modalities[key] === true) result[key] = true; } } - const limits = catalogSource.getLimits(provider, model); + const limits = source.getLimits(provider, model); if (limits) { if (limits.contextWindow > 0) result.contextWindow = limits.contextWindow; if (limits.maxOutput > 0) result.maxOutput = limits.maxOutput; @@ -413,12 +503,66 @@ function refine(base, provider, model) { return result; } +// Mirrors Command Code CLI `isKnownTextOnlyModel` (no image input). New models +// default to vision; only this denylist stays text-only. +const COMMANDCODE_TEXT_ONLY = new Set([ + "deepseek/deepseek-v4-pro", + "deepseek/deepseek-v4-flash", + "deepseek/deepseek-v4-flash-fast", + "zai-org/glm-5.3", + "zai-org/glm-5.2", + "zai-org/glm-5.2-fast", + "zai-org/glm-5.1", + "zai-org/glm-5", + "minimaxai/minimax-m2.7", + "minimax/minimax-m2.7-free", + "minimaxai/minimax-m2.5", + "xiaomi/mimo-v2.5-pro", + "qwen/qwen3.6-max-preview", + "qwen/qwen3.7-max", + "meituan/longcat-2.0:free", + "stepfun/step-3.5-flash", + "tencent/hy4-preview", + "tencent/hy3", + "tencent/hy3-paid", + "nvidia/nemotron-3-ultra-550b-a55b", + "poolside/laguna-s-2.1-free", + "inclusionai/ling-3.0-flash-free", + "inclusionai/ling-3.0-flash-sante:free", +]); + +function isCommandCodeTextOnly(model) { + const key = String(model || "").toLowerCase(); + if (COMMANDCODE_TEXT_ONLY.has(key)) return true; + for (const id of COMMANDCODE_TEXT_ONLY) { + const base = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id; + if (key === base || key.endsWith("/" + base)) return true; + } + return false; +} export function getCapabilitiesForModel(provider, model) { if (!model) return { ...DEFAULT_CAPABILITIES }; // Canonical exact lookup strips vendor prefix: "anthropic/claude-opus-4.7" -> "claude-opus-4.7". const baseModel = model.includes("/") ? model.split("/").pop() : model; + // CommandCode wire is /alpha/generate for every model. Family patterns + // (deepseek-v4 → thinkingFormat:deepseek, vision:false) must not win here. + if (provider === "commandcode" || provider === "cmc") { + const providerCaps = PROVIDER_CAPABILITIES.commandcode; + if (providerCaps?.[model]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[model] }; + if (providerCaps?.[baseModel]) return { ...DEFAULT_CAPABILITIES, ...providerCaps[baseModel] }; + return { + ...DEFAULT_CAPABILITIES, + reasoning: true, + thinkingFormat: "commandcode", + thinkingEffortSupported: true, + vision: !isCommandCodeTextOnly(model), + contextWindow: 1000000, + maxOutput: 384000, + }; + } + // 1. Provider-specific override if (provider) { const providerCaps = PROVIDER_CAPABILITIES[provider]; diff --git a/open-sse/providers/catalogOverride.js b/open-sse/providers/catalogOverride.js index 12914b7d..c34576d9 100644 --- a/open-sse/providers/catalogOverride.js +++ b/open-sse/providers/catalogOverride.js @@ -13,6 +13,11 @@ export const CATALOG_FILE = path.join(DATA_DIR, "model-catalog.json"); // Trimmed upstream catalog, read by the add-models skill (not by the router). export const CATALOG_RAW_FILE = path.join(DATA_DIR, "model-catalog-raw.json"); +// Schema of the file this module reads. The writer stamps it; a file carrying an +// older value predates provider-scoped modality keys, and its flat keys are not +// looked up here, so the sync rebuilds it instead of asking upstream for a 304. +export const CATALOG_VERSION = 2; + const EMPTY = { models: {}, providers: {} }; let cache = EMPTY; let cachedMtime = -1; @@ -45,14 +50,19 @@ function load() { return cache; } -// Modality is a property of the model itself — any gateway serving it inherits -// the same image/video/pdf support, so this is keyed by model id alone. -export function getCatalogModalities(model) { - return load().models[baseId(model)] || null; +// Modalities are recorded per gateway upstream, and gateways disagree about the +// same weights — some do not proxy images at all — so the key is provider + +// model, in the local provider id space, exactly like the limits below. Keying +// by model id alone made short ids collide across vendors: "auto", "free" and +// "efficient" are router modes in one catalog and model names in another, and a +// request to the router mode inherited a stranger's vision. +export function getCatalogModalities(provider, model) { + if (!provider) return null; + return load().models[`${provider}:${baseId(model)}`] || null; } -// Context and output limits are a property of the gateway, not the model: each -// one truncates differently, so these stay keyed by provider + model. +// Context and output limits are a property of the gateway too: each one +// truncates differently, so these stay keyed by provider + model. export function getCatalogLimits(provider, model) { const byProvider = provider && load().providers[provider]; if (!byProvider) return null; diff --git a/open-sse/providers/pricing.js b/open-sse/providers/pricing.js index 7cc302ff..d09452e2 100644 --- a/open-sse/providers/pricing.js +++ b/open-sse/providers/pricing.js @@ -53,6 +53,7 @@ export const MODEL_PRICING = { "gpt-5.6-luna": { input: 1.00, output: 6.00, cached: 0.10, reasoning: 6.00, cache_creation: 1.00 }, "gpt-5.6-terra": { input: 2.50, output: 15.00, cached: 0.25, reasoning: 15.00, cache_creation: 2.50 }, "gpt-5.6-sol": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 }, + "gpt-6-astra": { input: 5.00, output: 30.00, cached: 0.50, reasoning: 30.00, cache_creation: 5.00 }, "o1": { input: 15.00, output: 60.00, cached: 7.50, reasoning: 90.00, cache_creation: 15.00 }, "o1-mini": { input: 3.00, output: 12.00, cached: 1.50, reasoning: 18.00, cache_creation: 3.00 }, @@ -110,6 +111,8 @@ export const MODEL_PRICING = { "deepseek-v3.2-chat": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, "deepseek-v3.2-reasoner": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, "deepseek-v4-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, + "deepseek-v4.1-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, + "deepseek-flash": { input: 0.14, output: 0.28, cached: 0.0028, reasoning: 0.28, cache_creation: 0.14 }, "deepseek-v4-pro": { input: 0.435, output: 0.87, cached: 0.003625, reasoning: 0.87, cache_creation: 0.435 }, // === GLM === diff --git a/open-sse/providers/registry/antigravity.js b/open-sse/providers/registry/antigravity.js index ab1ee574..eba858e8 100644 --- a/open-sse/providers/registry/antigravity.js +++ b/open-sse/providers/registry/antigravity.js @@ -37,6 +37,7 @@ export default { }, usage: { quotaApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:fetchAvailableModels`, + quotaSummaryApiUrl: `${ANTIGRAVITY_IDE_BASE_URL}/v1internal:retrieveUserQuotaSummary`, loadProjectApiUrl: "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist", tokenUrl: "https://oauth2.googleapis.com/token", }, diff --git a/open-sse/providers/registry/api-airforce.js b/open-sse/providers/registry/api-airforce.js index 16ed8b3e..1f98bc20 100644 --- a/open-sse/providers/registry/api-airforce.js +++ b/open-sse/providers/registry/api-airforce.js @@ -20,6 +20,8 @@ export default { authModes: [ "apikey", ], + passthroughModels: true, + modelsFetcher: { url: "https://api.airforce/v1/models", type: "airforce-free" }, transport: { baseUrl: "https://api.airforce/v1/chat/completions", validateUrl: "https://api.airforce/v1/models", @@ -27,10 +29,11 @@ export default { "HTTP-Referer": "https://endpoint-proxy.local", "X-Title": "Endpoint Proxy", }, + forceStream: true, }, models: [ - { id: "anthropic/claude-3.7-sonnet", name: "Claude 3.7 Sonnet (Free)", contextLength: 200000 }, - { id: "moonshot/kimi-k2.6", name: "Kimi K2.6 (Free)", contextLength: 262144 }, - { id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash (Free)", contextLength: 1048576 }, + { id: "gpt-oss-120b", name: "GPT-OSS 120B (Free)", contextLength: 131072 }, + { id: "gpt-oss-20b", name: "GPT-OSS 20B (Free)", contextLength: 131072 }, + { id: "kimi-k2.7-code", name: "Kimi K2.7 Code (Free)", contextLength: 262144 }, ], }; diff --git a/open-sse/providers/registry/cline.js b/open-sse/providers/registry/cline.js index 6e40d6cf..340f5de3 100644 --- a/open-sse/providers/registry/cline.js +++ b/open-sse/providers/registry/cline.js @@ -23,6 +23,8 @@ export default { "HTTP-Referer": "https://cline.bot", "X-Title": "Cline", }, + // Non-stream chat completions come back wrapped in {"success":true,"data":{...}} + quirks: { clineEnvelope: true }, tokenUrl: "https://api.cline.bot/api/v1/auth/token", refreshUrl: "https://api.cline.bot/api/v1/auth/refresh", auth: { diff --git a/open-sse/providers/registry/clinepass.js b/open-sse/providers/registry/clinepass.js index 8f458b1f..4ff9436c 100644 --- a/open-sse/providers/registry/clinepass.js +++ b/open-sse/providers/registry/clinepass.js @@ -14,7 +14,10 @@ export default { }, }, category: "oauth", - authModes: ["oauth", "apikey"], + // ClinePass authenticates with a plain API key from app.cline.bot/settings/api-keys + // (category "apikey"). The OAuth extension flow used by Cline does not issue + // tokens that the ClinePass API consumer endpoint accepts (HTTP 401) — see #2333. + authModes: ["apikey", "oauth"], hasOAuth: true, transport: { baseUrl: "https://api.cline.bot/api/v1/chat/completions", @@ -22,6 +25,8 @@ export default { "HTTP-Referer": "https://cline.bot", "X-Title": "Cline", }, + // Non-stream chat completions come back wrapped in {"success":true,"data":{...}} + quirks: { clineEnvelope: true }, auth: { combined: true, header: "Authorization", diff --git a/open-sse/providers/registry/codebuddy-cn.js b/open-sse/providers/registry/codebuddy-cn.js index 33345b9f..672a05d8 100644 --- a/open-sse/providers/registry/codebuddy-cn.js +++ b/open-sse/providers/registry/codebuddy-cn.js @@ -47,27 +47,28 @@ export default { models: [ { id: "glm-5.2", name: "GLM-5.2" }, { id: "glm-5.1", name: "GLM-5.1" }, - { id: "glm-5.0-turbo", name: "GLM-5.0-Turbo" }, { id: "glm-5v-turbo", name: "GLM-5v-Turbo" }, { id: "minimax-m3", name: "MiniMax-M3" }, - { id: "minimax-m2.7", name: "MiniMax-M2.7" }, { id: "kimi-k2.7", name: "Kimi-K2.7-Code" }, { id: "kimi-k2.6", name: "Kimi-K2.6" }, - { id: "kimi-k2.5", name: "Kimi-K2.5" }, - // "-x" suffix = paid tier of the same model (free id rides the promo quota: - // hy3 free until 2026-08-31, hy4-preview until 2026-09-10). Server model table - // seen in client logs 2026-08-30; glm-5.0 / glm-4.7 removed (API 11102 dead). - { id: "hy3-preview", name: "Hy3 Preview" }, + // Catalog mirrors the server's product-config payload (the plugin fetches + // it from copilot.tencent.com). Models the server no longer publishes are + // removed even when the chat endpoint still answers them — the published + // list is the contract. Drop log: glm-5.0 / glm-4.7 and hy4-preview-x + // (endpoint returns 11102 "model service info not found"), plus + // glm-5.0-turbo / minimax-m2.7 / kimi-k2.5 / hy3-preview / + // deepseek-v3-2-volc (absent from the server list, though still answering + // 200) and hy3-x (paid tier, not used here). deepseek-v4-flash removed + // 2026-09: replaced server-side by deepseek-v4.1-flash (same low/high/ + // xhigh efforts; endpoint still answers 200 but the list is the contract). + // "-x" suffix = paid tier of the same model (free id rides the promo quota). { id: "hy3", name: "Hy3" }, - { id: "hy3-x", name: "Hy3 (Paid)" }, { id: "hy4-preview", name: "Hy4-Preview" }, - { id: "hy4-preview-x", name: "Hy4-Preview (Paid)" }, { id: "glm-5.3", name: "GLM-5.3" }, { id: "glm-5.3-flash", name: "GLM-5.3-Flash" }, { id: "kimi-k3-1", name: "Kimi-K3" }, { id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" }, - { id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" }, - { id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" }, + { id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" }, ], oauth: { baseUrl: "https://copilot.tencent.com", diff --git a/open-sse/providers/registry/codebuddy-intl.js b/open-sse/providers/registry/codebuddy-intl.js index eab1ce93..65e787d1 100644 --- a/open-sse/providers/registry/codebuddy-intl.js +++ b/open-sse/providers/registry/codebuddy-intl.js @@ -58,7 +58,9 @@ export default { { id: "kimi-k2.5", name: "Kimi-K2.5" }, { id: "hy3-preview", name: "Hy3 Preview" }, { id: "deepseek-v4-pro", name: "DeepSeek-V4-Pro" }, - { id: "deepseek-v4-flash", name: "DeepSeek-V4-Flash" }, + // deepseek-v4-flash replaced server-side by deepseek-v4.1-flash (same + // catalog as CN; the old endpoint still answers 200 but the list is the contract). + { id: "deepseek-v4.1-flash", name: "DeepSeek-V4.1-Flash" }, { id: "deepseek-v3-2-volc", name: "DeepSeek-V3.2" }, ], oauth: { diff --git a/open-sse/providers/registry/codex.js b/open-sse/providers/registry/codex.js index 6fc7501d..710cbc27 100644 --- a/open-sse/providers/registry/codex.js +++ b/open-sse/providers/registry/codex.js @@ -1,5 +1,9 @@ import { withCodexReviewModels } from "../models/helpers.js"; +// Codex CLI version seen by OpenAI's backend — single source for the Version / +// User-Agent identity headers. Bump when the installed codex CLI is upgraded. +const CODEX_CLI_VERSION = "0.154.0"; + export default { id: "codex", priority: 30, @@ -34,9 +38,10 @@ export default { baseUrl: "https://chatgpt.com/backend-api/codex/responses", format: "openai-responses", forceStream: true, + cliVersion: CODEX_CLI_VERSION, headers: { originator: "codex_cli_rs", - "User-Agent": "codex_cli_rs/0.136.0", + "User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}`, }, usage: { url: "https://chatgpt.com/backend-api/wham/usage", @@ -45,6 +50,7 @@ export default { }, }, models: [ + { id: "gpt-6-astra", name: "GPT 6.0 Astra" }, { id: "gpt-5.6-sol", name: "GPT 5.6 Sol" }, { id: "gpt-5.6-sol-review", name: "GPT 5.6 Sol Review", upstreamModelId: "gpt-5.6-sol", quotaFamily: "review" }, { id: "gpt-5.6-terra", name: "GPT 5.6 Terra" }, @@ -59,6 +65,17 @@ export default { { id: "gpt-5.4-mini-review", name: "GPT 5.4 Mini Review", upstreamModelId: "gpt-5.4-mini", quotaFamily: "review" }, { id: "gpt-5.3-codex-spark", name: "GPT 5.3 Codex Spark" }, { id: "gpt-5.3-codex-spark-review", name: "GPT 5.3 Codex Spark Review", upstreamModelId: "gpt-5.3-codex-spark", quotaFamily: "review" }, + // Codex CLI's auto-review virtual model. Unlike the "-review" variants above it is not derived + // from a base model, so it is forwarded verbatim instead of having "-review" stripped (#1398). + { id: "codex-auto-review", name: "Codex Auto Review", upstreamModelId: "codex-auto-review", quotaFamily: "review" }, + { id: "gpt-image-2.5", name: "GPT Image 2.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-2", name: "GPT Image 2", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-image-1.5", name: "GPT Image 1.5", capabilities: ["text2img","edit","multiImage"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-5.6-sol-image", name: "GPT 5.6 Sol Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-5.6-terra-image", name: "GPT 5.6 Terra Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, + { id: "gpt-5.6-luna-image", name: "GPT 5.6 Luna Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.5-image", name: "GPT 5.5 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.4-image", name: "GPT 5.4 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, { id: "gpt-5.3-image", name: "GPT 5.3 Image", capabilities: ["text2img","edit"], params: ["size","quality","background","image_detail","output_format"], kind: "image" }, diff --git a/open-sse/providers/registry/commandcode.js b/open-sse/providers/registry/commandcode.js index 3b21fbbc..f59aac04 100644 --- a/open-sse/providers/registry/commandcode.js +++ b/open-sse/providers/registry/commandcode.js @@ -40,4 +40,8 @@ export default { { id: "Qwen/Qwen3.6-Plus", name: "Qwen 3.6 Plus" }, { id: "stepfun/Step-3.5-Flash", name: "Step 3.5 Flash" }, ], + features: { + usage: true, + usageApikey: true, + }, }; diff --git a/open-sse/providers/registry/deepseek.js b/open-sse/providers/registry/deepseek.js index bb8015b0..997840e3 100644 --- a/open-sse/providers/registry/deepseek.js +++ b/open-sse/providers/registry/deepseek.js @@ -25,6 +25,21 @@ export default { reasoningInject: { scope: "all", }, + quirks: { + // DeepSeek's Anthropic-compatible endpoint + // (https://api.deepseek.com/anthropic/v1/messages) accepts ONLY the + // built-in web_search_* tools and rejects client-defined `custom` tools + // (MCP / Read / Bash / etc.) with HTTP 400 + // "tools[0]: unknown variant `custom`, expected + // `web_search_20250305` or `web_search_20260209`". + // + // Declaring this whitelist makes prepareClaudeRequest() forward only + // web_search_* tools and strip everything else before sending, so MCP / + // function tools are dropped instead of failing the whole request. + // DeepSeek's OpenAI-compatible transport is unaffected (targetFormat + // there is "openai", not "claude", so prepareClaudeRequest is not run). + claudeSupportedToolTypes: ["web_search_20250305", "web_search_20260209"], + }, }, // Multi-endpoint: pick the transport matching client sourceFormat to skip translation. transports: [ @@ -44,6 +59,7 @@ export default { { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro" }, { id: "deepseek-v4-pro-max", name: "DeepSeek V4 Pro Max", upstreamModelId: "deepseek-v4-pro" }, { id: "deepseek-v4-pro-none", name: "DeepSeek V4 Pro No Thinking", upstreamModelId: "deepseek-v4-pro" }, + { id: "deepseek-v4.1-flash", name: "DeepSeek V4.1 Flash" }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash" }, { id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)" }, { id: "deepseek-chat", name: "DeepSeek V3.2 Chat" }, diff --git a/open-sse/providers/registry/glm-cn.js b/open-sse/providers/registry/glm-cn.js index 73189464..7424547e 100644 --- a/open-sse/providers/registry/glm-cn.js +++ b/open-sse/providers/registry/glm-cn.js @@ -25,6 +25,7 @@ export default { { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" }, { id: "glm-5.2", name: "GLM 5.2" }, { id: "glm-5.1", name: "GLM 5.1" }, + { id: "glm-5-turbo", name: "GLM 5 Turbo" }, { id: "glm-5", name: "GLM 5" }, { id: "glm-4.7", name: "GLM-4.7" }, { id: "glm-4.6v", name: "GLM 4.6V (Vision)" }, diff --git a/open-sse/providers/registry/glm.js b/open-sse/providers/registry/glm.js index 6c5f0f6e..88f4c563 100644 --- a/open-sse/providers/registry/glm.js +++ b/open-sse/providers/registry/glm.js @@ -49,6 +49,7 @@ export default { { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)" }, { id: "glm-5.2", name: "GLM 5.2" }, { id: "glm-5.1", name: "GLM 5.1" }, + { id: "glm-5-turbo", name: "GLM 5 Turbo" }, { id: "glm-5", name: "GLM 5" }, { id: "glm-4.7", name: "GLM 4.7" }, { id: "glm-4.6v", name: "GLM 4.6V (Vision)" }, diff --git a/open-sse/providers/registry/minimax-cn.js b/open-sse/providers/registry/minimax-cn.js index 19aa3127..268a7344 100644 --- a/open-sse/providers/registry/minimax-cn.js +++ b/open-sse/providers/registry/minimax-cn.js @@ -22,6 +22,7 @@ export default { headers: { ...CLAUDE_API_HEADERS }, quirks: { dropOutputConfig: true, + requireClaudeToolType: true, }, reasoningInject: { scope: "all", diff --git a/open-sse/providers/registry/minimax.js b/open-sse/providers/registry/minimax.js index 47b82c89..66c00958 100644 --- a/open-sse/providers/registry/minimax.js +++ b/open-sse/providers/registry/minimax.js @@ -22,6 +22,7 @@ export default { headers: { ...CLAUDE_API_HEADERS }, quirks: { dropOutputConfig: true, + requireClaudeToolType: true, }, reasoningInject: { scope: "all", diff --git a/open-sse/providers/registry/ollama.js b/open-sse/providers/registry/ollama.js index 6965484a..605c4b89 100644 --- a/open-sse/providers/registry/ollama.js +++ b/open-sse/providers/registry/ollama.js @@ -30,6 +30,7 @@ export default { { id: "glm-4.7-flash", name: "GLM 4.7 Flash" }, { id: "qwen3.5", name: "Qwen3.5" }, { id: "minimax-m3", name: "MiniMax M3" }, + { id: "deepseek-v4.1-flash:cloud", name: "DeepSeek V4.1 Flash" }, ], serviceKinds: ["llm", "webFetch"], fetchConfig: { diff --git a/open-sse/providers/registry/openai.js b/open-sse/providers/registry/openai.js index 9a1ca57b..4a5f0eb1 100644 --- a/open-sse/providers/registry/openai.js +++ b/open-sse/providers/registry/openai.js @@ -57,6 +57,9 @@ export default { { id: "whisper-1", name: "Whisper 1", params: ["language","response_format","temperature","prompt"], kind: "stt" }, { id: "gpt-4o-transcribe", name: "GPT-4o Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" }, { id: "gpt-4o-mini-transcribe", name: "GPT-4o Mini Transcribe", params: ["language","response_format","temperature","prompt"], kind: "stt" }, + { id: "gpt-image-2.5", name: "GPT Image 2.5", params: ["n","size","quality","response_format"], kind: "image" }, + { id: "gpt-image-2.5-flare", name: "GPT Image 2.5 Flare", params: ["n","size","quality","response_format"], kind: "image" }, + { id: "gpt-image-2.5-sunburst", name: "GPT Image 2.5 Sunburst", params: ["n","size","quality","response_format"], kind: "image" }, { id: "gpt-image-1", name: "GPT Image 1", params: ["n","size","quality","response_format"], kind: "image" }, { id: "dall-e-3", name: "DALL-E 3", params: ["size","quality","style","response_format"], kind: "image" }, { id: "dall-e-2", name: "DALL-E 2", params: ["n","size","response_format"], kind: "image" }, diff --git a/open-sse/providers/registry/opencode-go.js b/open-sse/providers/registry/opencode-go.js index 0dad175f..d65d4c15 100644 --- a/open-sse/providers/registry/opencode-go.js +++ b/open-sse/providers/registry/opencode-go.js @@ -13,7 +13,7 @@ export default { textIcon: "OC", website: "https://opencode.ai/auth", notice: { - text: "OpenCode Go subscription: $5/mo (then 0/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.", + text: "OpenCode Go subscription: $5/mo (then 10/mo). Access to Kimi, GLM, Qwen, MiMo, MiniMax models.", apiKeyUrl: "https://opencode.ai/auth", }, }, @@ -21,6 +21,9 @@ export default { transport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions", headers: {}, + usage: { + url: "https://opencode.ai/zen/go/v1/usage", + }, }, // Multi-endpoint: pick the transport matching the client sourceFormat to skip // translation. Guarded per-model by `supportedFormats` (see chatCore) because @@ -30,22 +33,41 @@ export default { { format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } }, { format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } }, ], + // supportedFormats follow the endpoint table in https://opencode.ai/docs/go/ models: [ + { id: "deepseek-flash", name: "DeepSeek V4.1 Flash", supportedFormats: ["openai"] }, { id: "glm-5.3-flash", name: "GLM 5.3 Flash (Vision)", supportedFormats: ["openai"] }, + { id: "glm-5.3", name: "GLM 5.3", supportedFormats: ["openai"] }, { id: "glm-5.2", name: "GLM 5.2", supportedFormats: ["openai"] }, { id: "glm-5.1", name: "GLM 5.1", supportedFormats: ["openai"] }, { id: "kimi-k2.7-code", name: "Kimi K2.7 Code", supportedFormats: ["openai"] }, { id: "kimi-k2.6", name: "Kimi K2.6", supportedFormats: ["openai"] }, + { id: "kimi-k3", name: "Kimi K3", supportedFormats: ["openai"] }, { id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", supportedFormats: ["openai", "claude", "openai-responses"] }, { id: "deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision (Exp)", supportedFormats: ["openai", "claude", "openai-responses"] }, + { id: "longcat-2.0", name: "LongCat 2.0", supportedFormats: ["openai"] }, { id: "mimo-v2.5", name: "MiMo V2.5", supportedFormats: ["openai"] }, { id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", supportedFormats: ["openai"] }, { id: "minimax-m3", name: "MiniMax M3", supportedFormats: ["openai", "claude"] }, { id: "minimax-m2.7", name: "MiniMax M2.7", supportedFormats: ["openai", "claude"] }, { id: "minimax-m2.5", name: "MiniMax M2.5", supportedFormats: ["openai", "claude"] }, + { id: "qwen3.8-max", name: "Qwen 3.8 Max", supportedFormats: ["openai", "claude"] }, + { id: "qwen3.8-flash", name: "Qwen 3.8 Flash", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-max", name: "Qwen 3.7 Max", supportedFormats: ["openai", "claude"] }, { id: "qwen3.7-plus", name: "Qwen 3.7 Plus", supportedFormats: ["openai", "claude"] }, { id: "qwen3.6-plus", name: "Qwen 3.6 Plus", supportedFormats: ["openai", "claude"] }, + { id: "hy4-preview", name: "Hy4 Preview", supportedFormats: ["openai"] }, + { id: "hy3", name: "Hy3", supportedFormats: ["openai"] }, + // Served by /zen/go/v1/responses only — the responses-only entry forces chatCore + // past the sourceFormat-matched transports into translation (see chatCore guard). + { id: "grok-4.6", name: "Grok 4.6", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "gpt-5.6-luna", name: "GPT 5.6 Luna", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "muse-spark-1.2-contributor", name: "Muse Spark 1.2 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, + { id: "muse-spark-1.3-contributor", name: "Muse Spark 1.3 Contributor", targetFormat: "openai-responses", supportedFormats: ["openai-responses"] }, ], + features: { + usage: true, + usageApikey: true, + }, }; diff --git a/open-sse/providers/registry/opencode.js b/open-sse/providers/registry/opencode.js index 03e72eef..b0705066 100644 --- a/open-sse/providers/registry/opencode.js +++ b/open-sse/providers/registry/opencode.js @@ -17,13 +17,17 @@ export default { headers: { "x-opencode-client": "desktop", }, + forceStream: true, noAuth: true, + quirks: { + forceAutoToolChoiceModels: ["muse-spark-1.3-contributor-free"], + }, }, models: [ - // Muse Spark models are served by /zen/v1/responses; the rest stay on - // /chat/completions, so the format is declared per-model, not per-provider. + // Endpoint formats differ per model, so declare non-chat models explicitly. { id: "muse-spark-1.2-contributor-free", name: "Muse Spark 1.2 Contributor Free", targetFormat: "openai-responses" }, { id: "muse-spark-1.3-contributor-free", name: "Muse Spark 1.3 Contributor Free", targetFormat: "openai-responses" }, + { id: "union-alpha", name: "Union Alpha Free", targetFormat: "claude" }, ], modelsFetcher: { url: "https://opencode.ai/zen/v1/models", type: "opencode-free" }, passthroughModels: true, diff --git a/open-sse/providers/registry/openrouter.js b/open-sse/providers/registry/openrouter.js index a0df2a52..68a92857 100644 --- a/open-sse/providers/registry/openrouter.js +++ b/open-sse/providers/registry/openrouter.js @@ -40,8 +40,11 @@ export default { { id: "openai/gpt-image-1", name: "GPT Image 1 (via OpenRouter)", params: ["n","size","quality","response_format"], kind: "image" }, { id: "google/imagen-3.0-generate-002", name: "Imagen 3 (via OpenRouter)", params: ["n","size"], kind: "image" }, { id: "black-forest-labs/FLUX.1-schnell", name: "FLUX.1 Schnell (via OpenRouter)", params: ["n","size"], kind: "image" }, + { id: "google/veo-3.1", name: "Veo 3.1 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" }, + { id: "openai/sora-2-pro", name: "Sora 2 Pro (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" }, + { id: "bytedance/seedance-2.0", name: "Seedance 2.0 (via OpenRouter)", params: ["duration","aspect_ratio","resolution"], kind: "video" }, ], - serviceKinds: ["llm","embedding","tts","imageToText"], + serviceKinds: ["llm","embedding","tts","imageToText","video"], ttsConfig: { baseUrl: "https://openrouter.ai/api/v1/chat/completions", defaultModel: "openai/gpt-4o-mini-tts", @@ -57,6 +60,12 @@ export default { baseUrl: "https://openrouter.ai/api/v1/images/generations", headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"}, }, + // Async video jobs (POST /videos → { id, status }, GET /videos/{id} polls). + // Docs: https://openrouter.ai/docs/api/api-reference/videos + videoConfig: { + baseUrl: "https://openrouter.ai/api/v1/videos", + headers: {"HTTP-Referer":"https://endpoint-proxy.local","X-Title":"Endpoint Proxy"}, + }, modelsFetcher: { url: "https://openrouter.ai/api/v1/models", type: "openrouter-free" }, passthroughModels: true, }; diff --git a/open-sse/providers/registry/qoder.js b/open-sse/providers/registry/qoder.js index fe76fd72..042c83ae 100644 --- a/open-sse/providers/registry/qoder.js +++ b/open-sse/providers/registry/qoder.js @@ -30,12 +30,15 @@ export default { { id: "auto", name: "Auto" }, { id: "performance", name: "Performance" }, { id: "efficient", name: "Efficient" }, - { id: "qmodel_preview", name: "Qwen3.8-Max-Preview" }, + { id: "lite", name: "Lite" }, + { id: "qmodel_38max", name: "Qwen3.8-Max" }, { id: "qmodel_latest", name: "Qwen3.7-Max" }, { id: "qmodel", name: "Qwen3.7-Plus" }, + { id: "qfmodel", name: "Qwen3.8-Flash" }, { id: "kmodel_latest", name: "Kimi-K3" }, { id: "kmodel", name: "Kimi-K2.7-Code" }, - { id: "gm51model", name: "GLM-5.2" }, + { id: "gmodel", name: "GLM-5.3" }, + { id: "gfmodel", name: "GLM-5.3-Flash" }, { id: "dmodel", name: "DeepSeek-V4-Pro" }, { id: "dfmodel", name: "DeepSeek-V4-Flash" }, { id: "mmodel", name: "MiniMax-M3" }, diff --git a/open-sse/providers/registry/vertex.js b/open-sse/providers/registry/vertex.js index b8765de3..a1c60e60 100644 --- a/open-sse/providers/registry/vertex.js +++ b/open-sse/providers/registry/vertex.js @@ -27,6 +27,13 @@ export default { { id: "gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview" }, { id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview" }, { id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" }, + { id: "veo-3.1-generate-preview", name: "Veo 3.1 (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" }, + { id: "veo-3.1-fast-generate-preview", name: "Veo 3.1 Fast (Preview)", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" }, + { id: "veo-3.0-generate-001", name: "Veo 3", params: ["duration","aspect_ratio","resolution","negative_prompt","seed","storage_uri","generate_audio"], kind: "video" }, + { id: "veo-2.0-generate-001", name: "Veo 2", params: ["duration","aspect_ratio","negative_prompt","seed","storage_uri"], kind: "video" }, ], - serviceKinds: ["llm","imageToText"], + serviceKinds: ["llm","imageToText","video"], + // Veo via predictLongRunning + fetchPredictOperation (adapter: handlers/videoProviders/vertex.js). + // Docs: https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation + videoConfig: { baseUrl: "https://aiplatform.googleapis.com" }, }; diff --git a/open-sse/providers/registry/xiaomi-mimo.js b/open-sse/providers/registry/xiaomi-mimo.js index 49465f43..cb0139b3 100644 --- a/open-sse/providers/registry/xiaomi-mimo.js +++ b/open-sse/providers/registry/xiaomi-mimo.js @@ -1,11 +1,19 @@ import { CLAUDE_API_HEADERS } from "../shared.js"; +// Dual auth (same pattern as kimi): +// - API key (sk-...) → cloud API on api.xiaomimimo.com +// - Desktop account/OAuth → same cloud host, plus the Desktop-exclusive Preview +// models served by the account-service route on mimo-server-cn.xiaomimimo.com +// (authorized by a Xiaomi account session cookie, not the key). +// Endpoint is picked per model in the executor, same as opencode-go's /responses split. export default { id: "xiaomi-mimo", priority: 290, alias: "xiaomi-mimo", aliases: [ "mimo", + "mimo-desktop", + "xmd", ], uiAlias: "mimo", display: { @@ -16,9 +24,12 @@ export default { website: "https://xiaomimimo.com", notice: { apiKeyUrl: "https://platform.xiaomimimo.com/console/api-keys", + signupUrl: "https://mimo.xiaomimimo.com/desktop/invite/", }, }, - category: "apikey", + category: "oauth", + authModes: ["oauth", "apikey"], + hasOAuth: true, serviceKinds: ["llm", "tts"], transport: { baseUrl: "https://api.xiaomimimo.com/v1/chat/completions", @@ -39,6 +50,11 @@ export default { }, ], models: [ + // Desktop-exclusive — served by the account-service route, which only accepts + // OpenAI format, so supportedFormats pins them to the openai transport. + { id: "mimo-x-pro-preview", name: "MiMo-X-Pro-Preview", upstreamModelId: "xiaomi/mimo-x-pro-preview", supportedFormats: ["openai"] }, + { id: "mimo-x-flash-preview", name: "MiMo-X-Flash-Preview", upstreamModelId: "xiaomi/mimo-x-flash-preview", supportedFormats: ["openai"] }, + // Cloud API models (api.xiaomimimo.com/v1) { id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro" }, { id: "mimo-v2.5", name: "MiMo V2.5" }, { id: "mimo-v2-omni", name: "MiMo V2 Omni" }, @@ -51,4 +67,18 @@ export default { authHeader: "bearer", format: "xiaomi-mimo-tts", }, + features: { + usage: true, + usageApikey: true, + }, + // Custom OAuth — non-standard ECDH encrypted-callback flow. + // Handled by the Xiaomi MiMo OAuth service, not the generic PKCE pipeline. + oauth: { + custom: true, + authorizeUrl: "https://platform.xiaomimimo.com/authorize", + // The callback carries ?u= instead of ?code=. + // Decryption yields { uid, sk, url }. + callbackParam: "u", + kn: "mimocode", + }, }; diff --git a/open-sse/providers/registry/zed.js b/open-sse/providers/registry/zed.js index 9224cf95..a8ef14f1 100644 --- a/open-sse/providers/registry/zed.js +++ b/open-sse/providers/registry/zed.js @@ -1,10 +1,9 @@ // Zed provider — RSA keypair callback auth (NOT standard OAuth). export default { id: "zed", - priority: 10, + priority: 999, alias: "zd", uiAlias: "zd", - hidden: true, display: { name: "Zed", icon: "code", diff --git a/open-sse/providers/shared.js b/open-sse/providers/shared.js index 88488499..c0699c6f 100644 --- a/open-sse/providers/shared.js +++ b/open-sse/providers/shared.js @@ -62,8 +62,17 @@ const ANTHROPIC_BETA_BASE = [ const ANTHROPIC_BETA_HEAVY_AGENT = ["advanced-tool-use-2025-11-20", "effort-2025-11-24"]; // Heavy-agent beta flags are gated to opus/sonnet — cheaper models don't need them. -export function selectAnthropicBeta(model = "") { - const flags = [...ANTHROPIC_BETA_BASE]; +// `redact-thinking` asks Anthropic to return signature-only thinking blocks, which +// is right for clients that never render thinking but blanks the summaries a +// client explicitly requested with `thinking.display: "summarized"`. +const ANTHROPIC_BETA_REDACT_THINKING = "redact-thinking-2026-02-12"; + +export function wantsThinkingSummaries(body) { + return body?.thinking?.display === "summarized"; +} + +export function selectAnthropicBeta(model = "", body = null) { + const flags = ANTHROPIC_BETA_BASE.filter((flag) => flag !== ANTHROPIC_BETA_REDACT_THINKING || !wantsThinkingSummaries(body)); if (/^claude-(opus|sonnet)/.test(model)) flags.push(...ANTHROPIC_BETA_HEAVY_AGENT); return flags.join(","); } diff --git a/open-sse/providers/thinkingLevels.js b/open-sse/providers/thinkingLevels.js index b56896b1..94897ec7 100644 --- a/open-sse/providers/thinkingLevels.js +++ b/open-sse/providers/thinkingLevels.js @@ -26,6 +26,7 @@ const FORMAT_LEVELS = { qwen: L.base, kimi: L.levelMax, deepseek: L.hiMax, + commandcode: ["none", "low", "medium", "high", "xhigh", "max"], minimax: L.onOff, hunyuan: L.base, step: L.base, @@ -35,17 +36,29 @@ const CODEX_GPT_5_6_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh // Model-name pattern overrides (glob, first match wins) — more precise than format default. const PATTERN_THINKING = [ + { provider: "codex", pattern: "*gpt-6*", levels: CODEX_GPT_5_6_LEVELS }, { provider: "codex", pattern: "*gpt-5.6-sol*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-terra*", levels: [...CODEX_GPT_5_6_LEVELS, "ultra"] }, { provider: "codex", pattern: "*gpt-5.6-luna*", levels: CODEX_GPT_5_6_LEVELS }, { pattern: "*codex*", levels: ["low", "medium", "high", "xhigh"] }, // codex cannot disable thinking - // codebuddy-cn per-model effort sets — read off the client picker (server- - // delivered supportedEfforts), 2026-08-30. Gateway uses thinkingFormat "openai" - // but rejects levels outside each model's set. + // DeepSeek v4.* (Alibaba MaaS, probed live): effort low|medium|high|xhigh|max + // all 200 via output_config.effort; "none" is a 400 on the anthropic route + // (disable thinking instead). none kept for the picker = disable. + { pattern: "*deepseek-v4.*", levels: ["none", "low", "medium", "high", "xhigh", "max"] }, + // codebuddy-cn per-model effort sets — the server's product-config payload + // publishes `reasoning.supportedEfforts` per model. NOTE: the chat endpoint + // accepts any level you send (probed none/minimal/low/medium/high/xhigh/max + // → all 200), but values outside a model's supportedEfforts are silently + // clamped, so the declared set stays authoritative for the picker. Models + // that publish no supportedEfforts (glm-5.1 / glm-5v-turbo / kimi-k2.x / + // kimi-k3-1 / minimax-m3) fall through to the openai format default. { provider: "codebuddy-cn", pattern: "glm-5.3*", levels: ["low", "high", "max"] }, + { provider: "codebuddy-cn", pattern: "glm-5.2", levels: ["high", "xhigh"] }, { provider: "codebuddy-cn", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] }, { provider: "codebuddy-cn", pattern: "hy3*", levels: ["low", "high"] }, { provider: "codebuddy-cn", pattern: "hy4*", levels: ["high"] }, + // codebuddy-intl rides the same gateway catalog, so its deepseek levels match. + { provider: "codebuddy-intl", pattern: "deepseek-v4*", levels: ["low", "high", "xhigh"] }, ]; // Returns valid thinking levels for a model, or null when the model has no reasoning. diff --git a/open-sse/rtk/systemInject.js b/open-sse/rtk/systemInject.js index b60e15d3..b4e744f8 100644 --- a/open-sse/rtk/systemInject.js +++ b/open-sse/rtk/systemInject.js @@ -13,7 +13,7 @@ export function injectSystemPrompt(body, format, prompt) { if (!body || !prompt) return; if (typeof body !== "object") return; - // Kiro wire shape is unique (conversationState/systemPrompt) — handle directly. + // Kiro wire shape is unique (conversationState) — handle directly. if (isKiroBody(body) || format === FORMATS.KIRO) { injectKiroSystem(body, prompt); return; @@ -61,10 +61,13 @@ export function injectSystemPrompt(body, format, prompt) { function isKiroBody(body) { if (!body || typeof body !== "object") return false; - if (typeof body.systemPrompt !== "string") return false; const cs = body.conversationState; if (!cs || typeof cs !== "object") return false; - return Array.isArray(cs.history) || !!(cs.currentMessage && typeof cs.currentMessage === "object"); + // A top-level `systemPrompt` used to be the marker, but the Kiro translator no + // longer emits it (kiro.dev rejects the field), so gate on the turn shape. + const historyTurn = Array.isArray(cs.history) + && cs.history.some(it => it && (it.userInputMessage || it.assistantResponseMessage)); + return historyTurn || !!(cs.currentMessage && cs.currentMessage.userInputMessage); } // Exact idempotency: prompt present as its own SEP-delimited segment (or the @@ -258,80 +261,33 @@ function injectGeminiSystem(body, prompt) { } // ---- Kiro ---- -// Updates top-level systemPrompt and only the mirrored leading prefix of the -// first user history turn, else current user. next = old + SEP + prompt. -// Replace old leading prefix only; preserve time context and user tail. +// The prompt is appended to the first user turn's content — the same place the +// Kiro translator already mirrors the system text via its contentPrefix. +// +// A top-level `systemPrompt` is deliberately NOT written: the kiro.dev gateway +// answers any body carrying that field with +// 400 {"message":"Improperly formed request.","reason":"REQUEST_BODY_INVALID"} +// The translator stopped emitting it in v0.5.59, but this injector kept adding +// it back, so every kr/ model failed whenever an RTK prompt (caveman, ponytail) +// was active. function injectKiroSystem(body, prompt) { try { - let oldPrompt = typeof body.systemPrompt === "string" ? body.systemPrompt : ""; - // Repair path: a previous partial write left systemPrompt updated but user - // content still mirroring the pre-write prefix. Re-derive the effective old - // prefix from content so this pass converges instead of early-returning. - const cs0 = body.conversationState; - let firstUser0 = cs0 && Array.isArray(cs0.history) - ? (cs0.history.find(it => it && it.userInputMessage)?.userInputMessage ?? null) - : null; - if (!firstUser0 && cs0?.currentMessage?.userInputMessage) firstUser0 = cs0.currentMessage.userInputMessage; - - if (firstUser0 && typeof firstUser0.content === "string" && oldPrompt && !hasPrompt(oldPrompt, prompt)) { - const c0 = firstUser0.content; - if (c0 === oldPrompt || (c0.startsWith(oldPrompt) && !c0.startsWith(`${oldPrompt}${SEP}`))) { - // systemPrompt advanced past mirrored prefix → stale; treat as un-mirrored - oldPrompt = ""; - } - } - if (oldPrompt && hasPrompt(oldPrompt, prompt)) return; - const next = oldPrompt ? `${oldPrompt}${SEP}${prompt}` : prompt; - - // Atomicity: write user content first, then systemPrompt only if content - // write succeeded (or was a no-op). If systemPrompt write then fails, the - // repair heuristic above re-derives from content on retry — no permanent - // half-applied state. const cs = body.conversationState; let targetMsg = null; - try { - const hist = Array.isArray(cs?.history) ? cs.history : null; - if (hist) { - for (const item of hist) { - if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; } - } - } - if (!targetMsg && cs?.currentMessage?.userInputMessage) { - targetMsg = cs.currentMessage.userInputMessage; - } - } catch (_) { targetMsg = null; } - - let sysWritten = false; - try { body.systemPrompt = next; sysWritten = true; } catch (_) {} - - const applyContent = () => { - const content = typeof targetMsg.content === "string" ? targetMsg.content : ""; - if (oldPrompt === "") { - // Empty old prompt: prepend unless already at head (exact, not substring) - if (content.startsWith(prompt) || content.startsWith(next)) return; - const newContent = content ? `${next}${SEP}${content}` : next; - try { targetMsg.content = newContent; } catch (_) {} - return; - } - if (!content.startsWith(oldPrompt)) return; // not mirrored at head — leave alone - if (content.startsWith(next)) return; // already applied → idempotent - const tail = content.slice(oldPrompt.length); - try { targetMsg.content = `${next}${tail}`; } catch (_) {} - }; - - try { - if (targetMsg) applyContent(); - } catch (_) {} - if (sysWritten && targetMsg) { - // verify convergence: content should now start with next (or be un-mirrored) - let ok = false; - try { - const c = targetMsg.content; - ok = typeof c !== "string" || c.startsWith(next) || !c.startsWith(oldPrompt); - } catch (_) {} - if (!ok) { - try { body.systemPrompt = oldPrompt; } catch (_) {} // rollback + const hist = Array.isArray(cs?.history) ? cs.history : null; + if (hist) { + for (const item of hist) { + if (item && item.userInputMessage) { targetMsg = item.userInputMessage; break; } } } + if (!targetMsg && cs?.currentMessage?.userInputMessage) { + targetMsg = cs.currentMessage.userInputMessage; + } + if (!targetMsg) return; + + const content = typeof targetMsg.content === "string" ? targetMsg.content : ""; + const next = dedupStringAppend(content, prompt); + if (next === content) return; // already injected — idempotent across retries + try { targetMsg.content = next; } catch (_) { /* frozen/proxy fail-open */ } } catch (_) {} } diff --git a/open-sse/services/accountFallback.js b/open-sse/services/accountFallback.js index 8d280da4..766b9981 100644 --- a/open-sse/services/accountFallback.js +++ b/open-sse/services/accountFallback.js @@ -45,6 +45,20 @@ export function checkFallbackError(status, errorText, backoffLevel = 0) { } } + // Request-scoped client errors that matched no rule above: a 400 caused by the + // request itself (context overflow, malformed body, unsupported parameter) says + // nothing about the credential, so cooling the account down only removes a + // healthy connection from rotation. With a single connection it is worse: every + // later request in the window fails with a copy of this very error + // ("all 1 accounts locked for | lastError=[400]: ..."), which hides the + // real cause from the caller and makes unrelated sessions look like they hit the + // same limit. Hand the upstream error back for this request instead. + // Account-scoped statuses keep their rules above (401/402/403/404/429), and the + // text rules still win for rate-limit / quota / capacity wording. + if (status >= 400 && status < 500 && status !== 401 && status !== 402 && status !== 403 && status !== 429) { + return { shouldFallback: false, cooldownMs: 0 }; + } + // Default: transient cooldown for any unmatched error return { shouldFallback: true, cooldownMs: TRANSIENT_COOLDOWN_MS }; } diff --git a/open-sse/services/clinepassModels.js b/open-sse/services/clinepassModels.js index 8543999c..f656afd1 100644 --- a/open-sse/services/clinepassModels.js +++ b/open-sse/services/clinepassModels.js @@ -13,12 +13,10 @@ function buildModelListHeaders(token, isApiKey) { } /** - * Fetch ClinePass live model catalog from Cline's /models endpoint. - * - * @param {object} credentials - Connection credentials ({ accessToken, apiKey }) - * @returns {Promise<{ models: { id: string, name: string }[] } | null>} + * Internal: fetch the raw model list from Cline's /models endpoint. + * Returns the parsed array or null on any failure. */ -export async function resolveClinepassModels(credentials) { +async function fetchClineRawModels(credentials) { const isApiKey = Boolean(credentials?.apiKey); const token = isApiKey ? credentials.apiKey : credentials?.accessToken; if (!token) return null; @@ -39,19 +37,53 @@ export async function resolveClinepassModels(credentials) { const json = await response.json(); const rawList = Array.isArray(json) ? json : json?.data; - if (!Array.isArray(rawList)) return null; - - const models = rawList - .filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/")) - .map((m) => ({ - id: m.id, - name: m.name || m.id, - })); - - return models.length ? { models } : null; + return Array.isArray(rawList) ? rawList : null; } catch { return null; } finally { clearTimeout(timer); } } + +/** + * Fetch ClinePass live model catalog from Cline's /models endpoint. + * Returns only models with the cline-pass/ prefix. + * + * @param {object} credentials - Connection credentials ({ accessToken, apiKey }) + * @returns {Promise<{ models: { id: string, name: string }[] } | null>} + */ +export async function resolveClinepassModels(credentials) { + const rawList = await fetchClineRawModels(credentials); + if (!rawList) return null; + + const models = rawList + .filter((m) => typeof m?.id === "string" && m.id.startsWith("cline-pass/")) + .map((m) => ({ + id: m.id, + name: m.name || m.id, + })); + + return models.length ? { models } : null; +} + +/** + * Fetch Cline live model catalog from Cline's /models endpoint. + * Unlike resolveClinepassModels, this returns ALL models (including + * free-tier models like z-ai/glm-5.3-flash) without the cline-pass/ prefix filter. + * + * @param {object} credentials - Connection credentials ({ accessToken, apiKey }) + * @returns {Promise<{ models: { id: string, name: string }[] } | null>} + */ +export async function resolveClineModels(credentials) { + const rawList = await fetchClineRawModels(credentials); + if (!rawList) return null; + + const models = rawList + .filter((m) => typeof m?.id === "string" && m.id.trim() !== "") + .map((m) => ({ + id: m.id, + name: m.name || m.id, + })); + + return models.length ? { models } : null; +} diff --git a/open-sse/services/model.js b/open-sse/services/model.js index 5b88809c..15c50195 100644 --- a/open-sse/services/model.js +++ b/open-sse/services/model.js @@ -124,6 +124,8 @@ export async function getModelInfoCore(modelStr, aliasesOrGetter) { // Config-driven prefix → provider inference (first match wins, fallback "openai"). const MODEL_PREFIX_PROVIDERS = [ + // Codex CLI sends this bare virtual model for auto-review — keep it on OAuth Codex (#1398). + [/^codex-auto-review$/, "codex"], [/^claude-/, "anthropic"], [/^gemini-/, "gemini"], [/^gpt-/, "openai"], diff --git a/open-sse/services/projectId.js b/open-sse/services/projectId.js index 84ab5a2b..f582c4af 100644 --- a/open-sse/services/projectId.js +++ b/open-sse/services/projectId.js @@ -203,7 +203,8 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi const reqBody = { tierId: tierID, metadata: LOAD_CODE_ASSIST_METADATA }; const headers = provider === "antigravity" ? ANTIGRAVITY_LOAD_CODE_ASSIST_HEADERS : LOAD_CODE_ASSIST_HEADERS; - const MAX_ATTEMPTS = 5; + const MAX_ATTEMPTS = Number(process.env.ONBOARD_MAX_ATTEMPTS) || 2; + const BASE_RETRY_DELAY_MS = Number(process.env.ONBOARD_RETRY_DELAY_MS) || 12_000; for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) { // Bail out immediately if the connection was removed @@ -241,9 +242,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi throw new Error("onboardUser done but no project_id in response"); } - // Server not done yet – wait and retry + // Server not done yet – wait and retry with jitter + const jitter = Math.floor(Math.random() * 5000); console.log(`[ProjectId] Onboard attempt ${attempt}/${MAX_ATTEMPTS}: not done yet, waiting...`); - await new Promise(resolve => setTimeout(resolve, 2000)); + await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter)); } catch (error) { clearTimeout(timeoutId); @@ -256,9 +258,10 @@ async function onboardUser(accessToken, tierID, externalSignal, endpoints, provi console.warn(`[ProjectId] onboardUser failed after ${MAX_ATTEMPTS} attempts: ${error.message}`); return null; } - // Continue to next attempt instead of throwing (which would skip remaining retries) + // Wait with jitter before retrying + const jitter = Math.floor(Math.random() * 5000); console.warn(`[ProjectId] onboardUser attempt ${attempt} failed: ${error.message}, retrying...`); - await new Promise(resolve => setTimeout(resolve, 2000)); + await new Promise(resolve => setTimeout(resolve, BASE_RETRY_DELAY_MS + jitter)); } finally { clearTimeout(timeoutId); externalSignal?.removeEventListener("abort", forwardAbort); diff --git a/open-sse/services/qoderModels.js b/open-sse/services/qoderModels.js index 572931e5..e9a7879b 100644 --- a/open-sse/services/qoderModels.js +++ b/open-sse/services/qoderModels.js @@ -343,6 +343,30 @@ export async function resolveQoderModels(credentials, options = {}) { } } +/** + * Every model key the chat endpoint accepts for this credential: the IDE-visible + * models first, then catalog entries flagged `enable:false` (hidden in the IDE + * picker, e.g. by an account policy, but still served by agent_chat_generation — + * see fetchQoderCatalogRaw). /v1/models uses this so the advertised list matches + * what the router will actually route instead of collapsing to one or two keys. + */ +export function routableQoderModels(catalog) { + if (!catalog) return []; + const out = []; + const seen = new Set(); + for (const m of catalog.models || []) { + if (!m?.id || seen.has(m.id)) continue; + seen.add(m.id); + out.push({ id: m.id, name: m.name || m.id, hidden: false }); + } + for (const [key, cfg] of catalog.rawConfigs || []) { + if (!key || seen.has(key)) continue; + seen.add(key); + out.push({ id: key, name: cfg?.display_name || key, hidden: true }); + } + return out; +} + export function invalidateQoderCatalog(credentials) { if (!credentials) return; catalogCache.delete(cacheKey(credentials)); diff --git a/open-sse/services/thoughtSignatureStore.js b/open-sse/services/thoughtSignatureStore.js new file mode 100644 index 00000000..71fb94b8 --- /dev/null +++ b/open-sse/services/thoughtSignatureStore.js @@ -0,0 +1,199 @@ +import { makeKv } from "../../src/lib/db/helpers/kvStore.js"; + +const MAX_SIGNATURES = 2000; +const MAX_PERSISTED_SIGNATURES = 10_000; +const MEMORY_TTL_MS = 1000 * 60 * 60; // 1 hour +const PERSISTED_TTL_MS = 1000 * 60 * 60 * 24 * 7; // 7 days +const SCOPE = "gemini_thought_signatures"; + +const signatureKv = makeKv(SCOPE); +const memorySignatures = new Map(); +let pruneCounter = 0; + +/** + * Model family that produced / will consume a signature. Antigravity serves Gemini and Claude + * models behind the same API, and each backend only accepts its own signatures: a Claude + * signature replayed to Gemini fails with 400 "Corrupted thought signature." (and vice versa). + */ +export function signatureFamily(model) { + const m = typeof model === "string" ? model.toLowerCase() : ""; + if (!m) return null; + if (m.includes("claude")) return "claude"; + if (m.includes("gemini")) return "gemini"; + return m; +} + +// Entries stored before families were recorded (no `family`) stay usable for any model. +function isCompatible(entry, family) { + return !entry.family || !family || entry.family === family; +} + +function pruneMemoryExpired() { + const now = Date.now(); + for (const [key, value] of memorySignatures.entries()) { + if (value.expiresAt <= now) { + memorySignatures.delete(key); + } + } + + while (memorySignatures.size > MAX_SIGNATURES) { + const oldestKey = memorySignatures.keys().next().value; + if (!oldestKey) break; + memorySignatures.delete(oldestKey); + } +} + +async function maybePrunePersisted() { + pruneCounter++; + if (pruneCounter % 100 !== 0) return; + + try { + const all = await signatureKv.getAll(); + const keys = Object.keys(all); + const now = Date.now(); + const expiredKeys = []; + const valid = []; + + for (const k of keys) { + const entry = all[k]; + if (!entry || typeof entry.signature !== "string" || (entry.expiresAt && entry.expiresAt <= now)) { + expiredKeys.push(k); + } else { + valid.push({ key: k, createdAt: entry.createdAt || 0 }); + } + } + + for (const k of expiredKeys) { + await signatureKv.remove(k).catch(() => {}); + } + + if (valid.length > MAX_PERSISTED_SIGNATURES) { + valid.sort((a, b) => b.createdAt - a.createdAt); + const toRemove = valid.slice(MAX_PERSISTED_SIGNATURES); + for (const item of toRemove) { + await signatureKv.remove(item.key).catch(() => {}); + } + } + } catch { + // Fail-open + } +} + +/** + * Store a thought signature for a tool_call_id with optional sessionId namespace (RAM + SQLite async). + * `model` is the model that produced the signature; lookups for another model family skip it. + */ +export function storeGeminiThoughtSignature(toolCallId, signature, sessionId = null, model = null) { + if (typeof toolCallId !== "string" || !toolCallId) return; + if (typeof signature !== "string" || !signature) return; + + const now = Date.now(); + const family = signatureFamily(model); + pruneMemoryExpired(); + + const keys = []; + if (sessionId && typeof sessionId === "string") { + keys.push(`${sessionId}:${toolCallId}`); + } + keys.push(toolCallId); + + for (const k of keys) { + memorySignatures.set(k, { + signature, + family, + expiresAt: now + MEMORY_TTL_MS, + }); + + // Async persist to SQLite kv table without blocking + signatureKv.set(k, { + signature, + family, + createdAt: now, + expiresAt: now + PERSISTED_TTL_MS, + }).catch(() => {}); + } + + maybePrunePersisted().catch(() => {}); +} + +/** + * Retrieve a thought signature by tool_call_id (RAM first, then SQLite fallback). + * `model` is the target model; signatures produced by another model family are ignored. + */ +export async function getGeminiThoughtSignature(toolCallId, sessionId = null, model = null) { + if (typeof toolCallId !== "string" || !toolCallId) return null; + + const family = signatureFamily(model); + pruneMemoryExpired(); + + if (sessionId && typeof sessionId === "string") { + const sessionKey = `${sessionId}:${toolCallId}`; + const sessionEntry = memorySignatures.get(sessionKey); + if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) { + return sessionEntry.signature; + } + } + + const entry = memorySignatures.get(toolCallId); + if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) { + return entry.signature; + } + + try { + if (sessionId && typeof sessionId === "string") { + const sessionKey = `${sessionId}:${toolCallId}`; + const sessionRow = await signatureKv.get(sessionKey); + if (sessionRow && typeof sessionRow.signature === "string" && (!sessionRow.expiresAt || sessionRow.expiresAt > Date.now()) && isCompatible(sessionRow, family)) { + memorySignatures.set(sessionKey, { + signature: sessionRow.signature, + family: sessionRow.family || null, + expiresAt: Date.now() + MEMORY_TTL_MS, + }); + return sessionRow.signature; + } + } + + const row = await signatureKv.get(toolCallId); + if (row && typeof row.signature === "string") { + if (row.expiresAt && row.expiresAt <= Date.now()) { + signatureKv.remove(toolCallId).catch(() => {}); + return null; + } + if (!isCompatible(row, family)) return null; + memorySignatures.set(toolCallId, { + signature: row.signature, + family: row.family || null, + expiresAt: Date.now() + MEMORY_TTL_MS, + }); + return row.signature; + } + } catch { + // Fail-open + } + + return null; +} + +/** + * Synchronous get from RAM cache only (for sync translators). + * `model` is the target model; signatures produced by another model family are ignored. + */ +export function getGeminiThoughtSignatureSync(toolCallId, sessionId = null, model = null) { + if (typeof toolCallId !== "string" || !toolCallId) return null; + const family = signatureFamily(model); + pruneMemoryExpired(); + + if (sessionId && typeof sessionId === "string") { + const sessionKey = `${sessionId}:${toolCallId}`; + const sessionEntry = memorySignatures.get(sessionKey); + if (sessionEntry && sessionEntry.expiresAt > Date.now() && isCompatible(sessionEntry, family)) { + return sessionEntry.signature; + } + } + + const entry = memorySignatures.get(toolCallId); + if (entry && entry.expiresAt > Date.now() && isCompatible(entry, family)) { + return entry.signature; + } + return null; +} diff --git a/open-sse/services/tokenRefresh.js b/open-sse/services/tokenRefresh.js index dbf11ac2..ed1f4f38 100644 --- a/open-sse/services/tokenRefresh.js +++ b/open-sse/services/tokenRefresh.js @@ -148,6 +148,8 @@ const REFRESH_HANDLERS = { "codebuddy-intl": (c, log) => refreshCodebuddyIntlToken(c.refreshToken, log), trae: (c, log) => refreshTraeToken(c.refreshToken, c, log), cline: (c, log) => refreshClineToken(c.refreshToken, log), + // ClinePass shares Cline's WorkOS auth endpoints, so the same refresh works. + clinepass: (c, log) => refreshClineToken(c.refreshToken, log), zed: () => refreshZedToken(), windsurf: (c, log) => refreshWindsurfToken(c, log), // Kimi Code OAuth (merged into id `kimi`); legacy id still routes here diff --git a/open-sse/services/usage.js b/open-sse/services/usage.js index a2ed59b1..31321517 100644 --- a/open-sse/services/usage.js +++ b/open-sse/services/usage.js @@ -15,10 +15,13 @@ import { getGrokCliUsage } from "./usage/grok-cli.js"; import { getKimiUsage } from "./usage/kimi.js"; import { getDeepseekUsage } from "./usage/deepseek.js"; import { getFreebuffUsage } from "./usage/freebuff.js"; +import { getOpenCodeGoUsage } from "./usage/opencode-go.js"; import { getGroqUsage } from "./usage/groq.js"; import { getZedUsage } from "./usage/zed.js"; +import { getXiaomiMimoUsage } from "./usage/xiaomi-mimo.js"; import { resolveQoderCredentials } from "./qoderModels.js"; import { getGlmUsage } from "./usage/glm.js"; +import { getCommandCodeUsage } from "./usage/commandcode.js"; import { getIflowUsage, getOllamaUsage, @@ -56,10 +59,13 @@ const USAGE_HANDLERS = { "codebuddy-intl": (c) => getCodeBuddyIntlUsage(c.accessToken, c.apiKey, c.providerSpecificData, c.proxyOptions), "grok-cli": (c) => getGrokCliUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), kimi: (c) => getKimiUsage(c.accessToken, c.apiKey, c.proxyOptions, c.providerSpecificData), + "opencode-go": (c) => getOpenCodeGoUsage(c.apiKey, c.proxyOptions), deepseek: (c) => getDeepseekUsage(c.apiKey, c.proxyOptions), freebuff: (c) => getFreebuffUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), groq: (c) => getGroqUsage(c.apiKey, c.proxyOptions), zed: (c) => getZedUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), + "xiaomi-mimo": (c) => getXiaomiMimoUsage(c.accessToken, c.providerSpecificData, c.proxyOptions), + commandcode: (c) => getCommandCodeUsage(c.apiKey, c.proxyOptions), }; export async function getUsageForProvider(connection, proxyOptions = null, options = {}) { diff --git a/open-sse/services/usage/antigravity-weekly.js b/open-sse/services/usage/antigravity-weekly.js new file mode 100644 index 00000000..db92a224 --- /dev/null +++ b/open-sse/services/usage/antigravity-weekly.js @@ -0,0 +1,150 @@ +/** + * Antigravity weekly quota — best-effort retrieval from retrieveUserQuotaSummary. + * Failure never breaks existing per-model quota display. + */ + +import { U, parseResetTime, fetchWithTimeout } from "./shared.js"; +import { ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_IDE_VERSION } from "../../providers/shared.js"; + +// — Weekly quota summary config —————————————————————————————— +const WEEKLY_CONFIG = { + ...U("antigravity"), + userAgent: ANTIGRAVITY_IDE_USER_AGENT, +}; + +// — Cache: TTL + in-flight dedup per project ——————————————— +const WEEKLY_CACHE_TTL_MS = 180_000; // 3 minutes +const weeklyCache = new Map(); // cacheKey -> { result, expiresAt } | { promise } + +function cacheKey(accessToken, projectId) { + return `${accessToken}::${projectId || ""}`; +} + +// Exported for tests only +export function _clearWeeklyCache() { + weeklyCache.clear(); +} + +// — Group-name to stable key mapping —————————————————————— +const GROUP_MATCHERS = [ + { pattern: /gemini/i, key: "gemini_weekly", displayName: "Gemini (Weekly)" }, + { pattern: /claude|gpt/i, key: "claude_gpt_weekly", displayName: "Claude & GPT (Weekly)" }, +]; + +/** + * Parse a retrieveUserQuotaSummary response into normalized weekly quotas. + * Pure function — safe to unit-test without network. + * + * @param {Object|null} data Raw JSON response + * @returns {Object} e.g. { gemini_weekly: { used, total, ... }, claude_gpt_weekly: { ... } } + */ +export function parseWeeklyQuotaSummary(data) { + if (!data || typeof data !== "object") return {}; + + // Groups may live at data.groups or data.quotaSummary.groups + const groups = Array.isArray(data.groups) + ? data.groups + : Array.isArray(data.quotaSummary?.groups) + ? data.quotaSummary.groups + : null; + + if (!groups) return {}; + + const result = {}; + + for (const group of groups) { + if (!group || typeof group !== "object") continue; + const displayName = group.displayName || ""; + + const buckets = Array.isArray(group.buckets) ? group.buckets : []; + for (const bucket of buckets) { + if (!bucket || typeof bucket !== "object") continue; + + // Identify weekly buckets by checking bucketId + displayName for "weekly" + const bucketText = `${bucket.bucketId || ""} ${bucket.displayName || ""}`.toLowerCase(); + if (!bucketText.includes("weekly")) continue; + + // Skip disabled buckets + if (bucket.disabled === true) continue; + + const remainingFraction = Number(bucket.remainingFraction); + if (!Number.isFinite(remainingFraction)) continue; + + // Match group to a known family + for (const matcher of GROUP_MATCHERS) { + if (matcher.pattern.test(displayName)) { + const total = 1000; + const remaining = Math.round(total * remainingFraction); + const used = Math.max(0, total - remaining); + + result[matcher.key] = { + used, + total, + resetAt: parseResetTime(bucket.resetTime), + remainingPercentage: remainingFraction * 100, + unlimited: false, + displayName: matcher.displayName, + }; + break; // first matching bucket per family wins + } + } + } + } + + return result; +} + +/** + * Fetch weekly quota summary — cached, deduped, never throws. + */ +export async function fetchAntigravityWeeklyQuota(accessToken, projectId, proxyOptions = null) { + const key = cacheKey(accessToken, projectId); + + // Serve in-flight or cached + const hit = weeklyCache.get(key); + if (hit?.promise) return hit.promise; + if (hit && hit.expiresAt > Date.now()) return hit.result; + + const promise = (async () => { + try { + const url = WEEKLY_CONFIG.quotaSummaryApiUrl; + if (!url) return {}; + + const response = await fetchWithTimeout(url, { + method: "POST", + headers: { + "Authorization": `Bearer ${accessToken}`, + "User-Agent": WEEKLY_CONFIG.userAgent, + "Content-Type": "application/json", + "X-Client-Name": "antigravity", + "X-Client-Version": ANTIGRAVITY_IDE_VERSION, + }, + body: JSON.stringify({ + ...(projectId ? { project: projectId } : {}), + }), + }, 10000, proxyOptions); + + if (!response.ok) return {}; + + const data = await response.json(); + return parseWeeklyQuotaSummary(data); + } catch { + return {}; + } + })(); + + weeklyCache.set(key, { promise }); + + try { + const result = await promise; + if (result && Object.keys(result).length > 0) { + weeklyCache.set(key, { result, expiresAt: Date.now() + WEEKLY_CACHE_TTL_MS }); + } else { + weeklyCache.delete(key); + } + return result; + } catch { + weeklyCache.delete(key); + return {}; + } +} diff --git a/open-sse/services/usage/claude.js b/open-sse/services/usage/claude.js index ce3e01a0..d93a0682 100644 --- a/open-sse/services/usage/claude.js +++ b/open-sse/services/usage/claude.js @@ -110,6 +110,22 @@ async function fetchClaudeUsageRaw(accessToken, proxyOptions = null) { } } + // Model-scoped weekly limits (e.g. Fable) arrive in limits[], not as + // seven_day_* keys: { kind: "weekly_scoped", percent, resets_at, + // scope: { model: { display_name: "Fable" } } }. No limits entry means + // the account has no such window — omit the row, never fabricate one. + if (Array.isArray(data.limits)) { + for (const limit of data.limits) { + if (limit?.kind !== "weekly_scoped") continue; + const modelName = String(limit?.scope?.model?.display_name || "").trim().toLowerCase(); + if (!modelName || typeof limit.percent !== "number") continue; + quotas[`weekly ${modelName} (7d)`] = createQuotaObject({ + utilization: Math.max(0, Math.min(100, limit.percent)), + resets_at: limit.resets_at, + }); + } + } + return { plan: "Claude Code", extraUsage: data.extra_usage ?? null, diff --git a/open-sse/services/usage/codex.js b/open-sse/services/usage/codex.js index 64d3cbbc..67d91c4b 100644 --- a/open-sse/services/usage/codex.js +++ b/open-sse/services/usage/codex.js @@ -21,6 +21,13 @@ function toIsoDate(value) { return Number.isFinite(time) ? date.toISOString() : null; } +function errorMessage(value, fallback) { + if (!value) return fallback; + if (typeof value === "string") return value; + if (typeof value.message === "string") return value.message; + return JSON.stringify(value); +} + function getCodexAccountId(providerSpecificData) { return providerSpecificData?.workspaceId || providerSpecificData?.accountId || providerSpecificData?.chatgptAccountId || null; } @@ -162,7 +169,7 @@ export async function getCodexRateLimitResetCredits(accessToken, proxyOptions = } if (!response.ok) { - const message = data?.message || data?.error || data?.detail || `Codex reset credits API unavailable (${response.status}).`; + const message = errorMessage(data?.message || data?.error || data?.detail, `Codex reset credits API unavailable (${response.status}).`); throw new Error(message); } diff --git a/open-sse/services/usage/commandcode.js b/open-sse/services/usage/commandcode.js new file mode 100644 index 00000000..7c62480c --- /dev/null +++ b/open-sse/services/usage/commandcode.js @@ -0,0 +1,134 @@ +/** + * Command Code usage — billing credits + 5h/weekly rate windows. + * Mirrors ~/cc-usage.mjs: whoami → credits + subscriptions. + */ + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { parseResetTime, toFiniteNumber } from "./shared.js"; + +const BASE = (process.env.COMMAND_CODE_API_BASE_URL || "https://api.commandcode.ai").replace(/\/$/, ""); + +const PLAN_NAMES = { + "individual-go": "Go", + "individual-goat": "GOAT", + "individual-pro": "Pro", + "individual-pro-v1": "Pro", + "individual-provider": "Provider", + "individual-max": "Max", + "individual-ultra": "Ultra", + "teams-pro": "Teams Pro", +}; + +const PLAN_CAPS = { + "individual-go": 10, + "individual-goat": 70, + "individual-pro": 30, + "individual-pro-v1": 80, + "individual-provider": 15, + "individual-max": 150, + "individual-ultra": 300, + "teams-pro": 40, +}; + +function qs(route, params) { + const s = new URLSearchParams( + Object.entries(params || {}).filter(([, v]) => v != null), + ).toString(); + return s ? `${route}?${s}` : route; +} + +function windowQuota(win) { + if (!win || typeof win !== "object") return null; + const used = toFiniteNumber(win.used, 0); + const total = toFiniteNumber(win.cap, 0); + if (total <= 0 && used <= 0) return null; + return { + used, + total, + remaining: Math.max(0, total - used), + unlimited: false, + resetAt: parseResetTime(win.resetAt), + }; +} + +/** + * @param {string|null|undefined} apiKey + * @param {object|null} proxyOptions + */ +export async function getCommandCodeUsage(apiKey, proxyOptions = null) { + if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) { + return { message: "Command Code API key not available. Add a key to view usage." }; + } + + const headers = { + Authorization: `Bearer ${apiKey.trim()}`, + Accept: "application/json", + }; + + const get = async (route) => { + const response = await proxyAwareFetch( + BASE + route, + { method: "GET", headers }, + proxyOptions, + ); + return response; + }; + + try { + const whoamiRes = await get(qs("/alpha/whoami", { limits: "1" })); + if (whoamiRes.status === 401 || whoamiRes.status === 403) { + return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." }; + } + if (!whoamiRes.ok) { + return { plan: "Command Code", message: `Command Code usage API error (${whoamiRes.status})` }; + } + const whoami = await whoamiRes.json().catch(() => ({})); + const orgId = whoami?.org?.id ?? null; + + const [creditsRes, subsRes] = await Promise.all([ + get(qs("/alpha/billing/credits", { orgId })), + get(qs("/alpha/billing/subscriptions", { orgId })), + ]); + + if (creditsRes.status === 401 || creditsRes.status === 403 || subsRes.status === 401 || subsRes.status === 403) { + return { plan: "Command Code", message: "Command Code authentication failed. Check the API key." }; + } + if (!creditsRes.ok) { + return { plan: "Command Code", message: `Command Code credits API error (${creditsRes.status})` }; + } + if (!subsRes.ok) { + return { plan: "Command Code", message: `Command Code subscriptions API error (${subsRes.status})` }; + } + + const creditsBody = await creditsRes.json().catch(() => ({})); + const subsBody = await subsRes.json().catch(() => ({})); + const planId = subsBody?.data?.planId ?? null; + const plan = (planId && PLAN_NAMES[planId]) || planId || "Command Code"; + const cap = planId ? (PLAN_CAPS[planId] || 0) : 0; + const c = creditsBody?.credits || {}; + const remaining = + toFiniteNumber(c.monthlyCredits, 0) + + toFiniteNumber(c.purchasedCredits, 0) + + toFiniteNumber(c.freeCredits, 0); + const used = cap > 0 ? Math.max(0, cap - remaining) : 0; + const total = cap > 0 ? cap : remaining; + + const quotas = {}; + quotas.Credits = { + used, + total, + remaining, + unlimited: cap <= 0, + resetAt: parseResetTime(subsBody?.data?.currentPeriodEnd), + }; + + const fiveHour = windowQuota(creditsBody?.windowLimits?.fiveHour); + if (fiveHour) quotas["Session (5h)"] = fiveHour; + const weekly = windowQuota(creditsBody?.windowLimits?.weekly); + if (weekly) quotas.Weekly = weekly; + + return { plan, quotas }; + } catch (error) { + return { message: `Command Code error: ${error.message}` }; + } +} diff --git a/open-sse/services/usage/deepseek.js b/open-sse/services/usage/deepseek.js index cb70a40d..9d2ed2ac 100644 --- a/open-sse/services/usage/deepseek.js +++ b/open-sse/services/usage/deepseek.js @@ -91,14 +91,15 @@ export async function getDeepseekUsage(apiKey = null, proxyOptions = null) { const quotas = {}; for (const b of balances) { const total = Math.max(0, b.totalBalance); - // Credit pot: show full remaining against current balance; never set absolute - // `remaining` — QuotaTable treats it as a 0–100 percentage. + // Credit balance: show as "Credit: $X.XX USD" not a usage quota quotas[`Balance (${b.currency})`] = { used: 0, total, remainingPercentage: total > 0 ? 100 : 0, resetAt: null, - unlimited: total > 0, + unlimited: false, + isCreditBalance: true, + currency: b.currency, }; } diff --git a/open-sse/services/usage/google.js b/open-sse/services/usage/google.js index 9afbe4ca..736722a7 100644 --- a/open-sse/services/usage/google.js +++ b/open-sse/services/usage/google.js @@ -5,6 +5,7 @@ import { CLIENT_METADATA } from "../../config/appConstants.js"; import { ANTIGRAVITY_IDE_USER_AGENT, ANTIGRAVITY_IDE_VERSION, ANTIGRAVITY_OAUTH_CLIENT } from "../../providers/shared.js"; import { U, parseResetTime, normalizeCloudCodeProjectId, fetchWithTimeout } from "./shared.js"; +import { fetchAntigravityWeeklyQuota } from "./antigravity-weekly.js"; // Antigravity API config (from Quotio) — urls from registry, oauth client + dynamic UA kept here const ANTIGRAVITY_CONFIG = { @@ -157,8 +158,15 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro const data = await response.json(); const quotas = {}; - // Parse model quotas (inspired by vscode-antigravity-cockpit) - if (data.models) { + // Detect tier: free-tier accounts only have weekly quotas (no separate 5h window). + // On free-tier, fetchAvailableModels returns misleading per-model quota info + // (missing remainingFraction defaults to 0, or reflects the weekly limit not a 5h window). + const paidTierId = subscriptionInfo?.paidTier?.id; + const isFreeTier = !paidTierId || paidTierId === "free-tier"; + + // Parse model quotas only for paid-tier accounts. + // Free-tier accounts skip this — their only meaningful quota is the weekly limit. + if (!isFreeTier && data.models) { // Filter only recommended/important models (must match PROVIDER_MODELS ag ids) const importantModels = [ 'gemini-3.8-flash-high', @@ -212,6 +220,56 @@ export async function getAntigravityUsage(accessToken, providerSpecificData, pro } } + // Best-effort weekly quota overlay — never blocks or breaks per-model results + try { + const weeklyQuotas = await fetchAntigravityWeeklyQuota( + accessToken, + projectId, + proxyOptions + ); + + // Reconcile weekly quota against model family status: + // If every model in a family is locked/exhausted (remainingPercentage === 0) + // until a future reset time, the weekly limit cannot be 100% available. + // On Google's Free Starter tier, retrieveUserQuotaSummary buggily reports + // remainingFraction: 1 even after the starter quota is depleted and all models 429. + const entries = Object.entries(quotas); + const geminiModels = entries.filter(([k]) => k.startsWith("gemini-") && !k.includes("image")); + const claudeModels = entries.filter(([k]) => k.startsWith("claude-")); + + if (weeklyQuotas.gemini_weekly && geminiModels.length > 0) { + const allGeminiExhausted = geminiModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0); + if (allGeminiExhausted && weeklyQuotas.gemini_weekly.remainingPercentage > 0) { + const maxResetAt = geminiModels.reduce((max, [, q]) => + !max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null + ); + weeklyQuotas.gemini_weekly.used = weeklyQuotas.gemini_weekly.total; + weeklyQuotas.gemini_weekly.remainingPercentage = 0; + if (maxResetAt) { + weeklyQuotas.gemini_weekly.resetAt = maxResetAt; + } + } + } + + if (weeklyQuotas.claude_gpt_weekly && claudeModels.length > 0) { + const allClaudeExhausted = claudeModels.every(([, q]) => (q.remainingPercentage ?? 0) === 0); + if (allClaudeExhausted && weeklyQuotas.claude_gpt_weekly.remainingPercentage > 0) { + const maxResetAt = claudeModels.reduce((max, [, q]) => + !max || (q.resetAt && new Date(q.resetAt) > new Date(max)) ? q.resetAt : max, null + ); + weeklyQuotas.claude_gpt_weekly.used = weeklyQuotas.claude_gpt_weekly.total; + weeklyQuotas.claude_gpt_weekly.remainingPercentage = 0; + if (maxResetAt) { + weeklyQuotas.claude_gpt_weekly.resetAt = maxResetAt; + } + } + } + + Object.assign(quotas, weeklyQuotas); + } catch { + // Silently ignore — weekly is best-effort + } + return { plan: subscriptionInfo?.currentTier?.name || "Unknown", quotas, diff --git a/open-sse/services/usage/opencode-go.js b/open-sse/services/usage/opencode-go.js new file mode 100644 index 00000000..2d352498 --- /dev/null +++ b/open-sse/services/usage/opencode-go.js @@ -0,0 +1,107 @@ +/** + * OpenCode Go usage — GET https://opencode.ai/zen/go/v1/usage + * Auth: Bearer + */ + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { parseResetTime, toFiniteNumber, U } from "./shared.js"; + +const USAGE_URL = U("opencode-go").url; +const QUOTA_NAMES = { + rolling: "Rolling", + weekly: "Weekly", + monthly: "Monthly", +}; + +function parsePercent(value) { + if (typeof value === "number" && Number.isFinite(value)) return value; + if (typeof value === "string" && value.trim()) { + const parsed = Number(value); + if (Number.isFinite(parsed)) return parsed; + } + return null; +} + +export async function getOpenCodeGoUsage(apiKey = null, proxyOptions = null) { + if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) { + return { + message: "OpenCode Go API key not available. Add a key to view usage.", + }; + } + + try { + const response = await proxyAwareFetch( + USAGE_URL, + { + method: "GET", + headers: { + Authorization: `Bearer ${apiKey.trim()}`, + Accept: "application/json", + }, + }, + proxyOptions, + ); + + if (response.status === 401) { + return { + plan: "OpenCode Go", + message: "OpenCode Go authentication failed. Check the API key.", + }; + } + + if (response.status === 403) { + const error = await response.json().catch(() => null); + const subscriptionRequired = error?.error?.type === "EntitlementError"; + return { + plan: "OpenCode Go", + message: subscriptionRequired + ? "OpenCode Go subscription required for this API key." + : "OpenCode Go access forbidden for this API key.", + }; + } + + if (!response.ok) { + return { + plan: "OpenCode Go", + message: `OpenCode Go usage API error (${response.status}).`, + }; + } + + const data = await response.json().catch(() => null); + if (!data?.usage || typeof data.usage !== "object") { + return { + plan: "OpenCode Go", + message: "OpenCode Go usage response did not contain quota data.", + }; + } + + const quotas = {}; + for (const [period, name] of Object.entries(QUOTA_NAMES)) { + const quota = data.usage[period]; + if (!quota || typeof quota !== "object") continue; + const percent = parsePercent(quota.percent); + if (percent === null) continue; + const used = Math.max(0, Math.min(100, toFiniteNumber(percent, 0))); + quotas[name] = { + used, + total: 100, + remaining: 100 - used, + remainingPercentage: 100 - used, + resetAt: parseResetTime(quota.resetsAt), + unlimited: false, + }; + } + + + if (Object.keys(quotas).length === 0) { + return { + plan: "OpenCode Go", + message: "OpenCode Go usage response did not contain valid quota data.", + }; + } + + return { plan: "OpenCode Go", quotas }; + } catch (error) { + return { message: `OpenCode Go error: ${error.message}` }; + } +} diff --git a/open-sse/services/usage/xiaomi-mimo.js b/open-sse/services/usage/xiaomi-mimo.js new file mode 100644 index 00000000..9d42fe37 --- /dev/null +++ b/open-sse/services/usage/xiaomi-mimo.js @@ -0,0 +1,125 @@ +/** + * Xiaomi MiMo usage — weekly quota from the Xiaomi account session. + * + * Primary path: GET {mimo-server}/api/user/usage authorized by the account-session + * cookie (see shared/mimoAccount.js). Response: { code: 0, data: { percent (remaining + * %), resetDate, resetAt } }. + * + * Fallback: the sk- API key cannot read the quota, so when no account session is + * available we surface a graceful message instead of failing. + */ + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { getMimoAccountUsage } from "../../shared/mimoAccount.js"; + +const USAGE_URL = "https://aistudio.xiaomimimo.com/open-apis/v1/user/usage"; + +/** + * @param {string|null|undefined} accessToken - sk- API key + * @param {object|null} providerSpecificData - may contain mimoPassToken, uid, etc. + * @param {object|null} proxyOptions + */ +export async function getXiaomiMimoUsage(accessToken = null, providerSpecificData = null, proxyOptions = null) { + // Preferred path: the weekly quota comes from the account service session + // (mimo-server /api/user/usage), which the sk- key cannot reach. The session is + // derived from MiMo Desktop's persisted passToken via the SSO/sts handshake. + const account = await getMimoAccountUsage(providerSpecificData, proxyOptions); + if (typeof account.percent === "number" && Number.isFinite(account.percent)) { + const remaining = Math.max(0, Math.min(100, Math.round(account.percent))); + const used = 100 - remaining; + let resetAt = null; + if (typeof account.resetAt === "number" && account.resetAt > 0) { + resetAt = new Date(account.resetAt * 1000).toISOString(); + } else if (typeof account.resetDate === "string") { + const parsed = new Date(`${account.resetDate}T00:00:00Z`); + if (!Number.isNaN(parsed.getTime())) resetAt = parsed.toISOString(); + } + return { + plan: "Xiaomi MiMo Desktop", + quotas: { + Weekly: { used, total: 100, remainingPercentage: remaining, resetAt, unlimited: false }, + }, + }; + } + + // Fallback: no account session available (Desktop never logged in, or its cookie + // store is locked). The sk- key cannot read the quota, so surface a clear message. + const key = accessToken || providerSpecificData?.apiKey; + if (!key || typeof key !== "string" || !key.trim()) { + return { message: "Xiaomi MiMo Desktop not connected. Add credentials to view usage." }; + } + + try { + const response = await proxyAwareFetch( + USAGE_URL, + { + method: "GET", + headers: { + Authorization: `Bearer ${key.trim()}`, + "X-Mimo-Source": "mimocode-cli", + Accept: "application/json", + }, + signal: AbortSignal.timeout(10000), + }, + proxyOptions, + ); + + if (response.status === 401) { + return { + plan: "Xiaomi MiMo Desktop", + message: "Weekly quota requires Xiaomi account session. API key alone is insufficient.", + }; + } + + if (!response.ok) { + return { + plan: "Xiaomi MiMo Desktop", + message: `Usage API error (${response.status})`, + }; + } + + const data = await response.json().catch(() => null); + if (!data || data.code !== 0 || !data.data) { + return { + plan: "Xiaomi MiMo Desktop", + message: "Usage endpoint returned unexpected response.", + }; + } + + const { percent, resetDate } = data.data; + if (typeof percent !== "number" || !Number.isFinite(percent)) { + return { + plan: "Xiaomi MiMo Desktop", + message: "Usage data missing percent field.", + }; + } + + // percent = remaining percentage (e.g. 94 means 94% remaining) + const remaining = Math.max(0, Math.min(100, Math.round(percent))); + const used = 100 - remaining; + + // Parse resetDate — expected format "2026-09-16" + let resetAt = null; + if (resetDate && typeof resetDate === "string") { + const parsed = new Date(`${resetDate}T00:00:00Z`); + if (!Number.isNaN(parsed.getTime())) { + resetAt = parsed.toISOString(); + } + } + + return { + plan: "Xiaomi MiMo Desktop", + quotas: { + Weekly: { + used, + total: 100, + remainingPercentage: remaining, + resetAt, + unlimited: false, + }, + }, + }; + } catch (error) { + return { message: `Xiaomi MiMo Desktop usage error: ${error.message}` }; + } +} diff --git a/open-sse/shared/clineAuth.js b/open-sse/shared/clineAuth.js index a7e7e407..c79f69d0 100644 --- a/open-sse/shared/clineAuth.js +++ b/open-sse/shared/clineAuth.js @@ -10,7 +10,14 @@ export function getClineAccessToken(token) { if (typeof token !== "string") return ""; const trimmed = token.trim(); if (!trimmed) return ""; - return trimmed.startsWith("workos:") ? trimmed : `workos:${trimmed}`; + if (trimmed.toLowerCase().startsWith("workos:")) return trimmed; + // Cline OAuth access tokens are WorkOS JWTs (base64url `eyJ…` header). + // ClinePass API keys (category "apikey", e.g. `clp_…`) are NOT JWTs and must + // be sent verbatim — prefixing them with `workos:` makes the Cline API reject + // the request with HTTP 401 ("Please make sure you're using the latest + // version of Cline and re-authenticate your Cline account."). + const isWorkOsJwt = /^eyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+/.test(trimmed); + return isWorkOsJwt ? `workos:${trimmed}` : trimmed; } export function getClineAuthorizationHeader(token) { diff --git a/open-sse/shared/clineEnvelope.js b/open-sse/shared/clineEnvelope.js new file mode 100644 index 00000000..3873b746 --- /dev/null +++ b/open-sse/shared/clineEnvelope.js @@ -0,0 +1,19 @@ +import { PROVIDERS } from "../providers/index.js"; + +/** + * Unwrap Cline's non-stream envelope: {"success":true,"data":{...choices...}}. + * + * Scoped to providers opting in via `transport.quirks.clineEnvelope` so no other + * provider's body is ever rewritten. The error envelope ({"success":false,...}) + * never matches and passes through untouched. + * + * @param {object} body - Parsed upstream response body + * @param {string} provider - Provider id or alias + * @returns {object} The inner `data` object, or `body` unchanged + */ +export function unwrapClineEnvelope(body, provider) { + if (!provider || !PROVIDERS[provider]?.quirks?.clineEnvelope) return body; + const { success, data } = body || {}; + if (success !== true || !data || typeof data !== "object" || Array.isArray(data)) return body; + return data; +} diff --git a/open-sse/shared/mimoAccount.js b/open-sse/shared/mimoAccount.js new file mode 100644 index 00000000..996f38df --- /dev/null +++ b/open-sse/shared/mimoAccount.js @@ -0,0 +1,264 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import crypto from "node:crypto"; +import { proxyAwareFetch } from "../utils/proxyFetch.js"; + +/** + * Xiaomi MiMo account-session helpers (used for weekly quota). + * + * The weekly quota endpoint lives on the account service domain and is authorized + * by an account session cookie, NOT the sk- API key. Acquiring that cookie mirrors + * MiMo Desktop: a passToken (persisted in Desktop's cookie store) is exchanged via + * the passportapi SSO, then authorized for the `mimopc` service, and finally stamped + * by the mimo-server /api/sts callback into a `serviceToken` cookie. + * + * Flow (verified against MiMo Desktop traffic): + * 1. GET {api}/api/user/xiaomi/me -> 302 to account SSO (sid=mimopc) + * 2. GET account /pass/serviceLogin?sid=passportapi&_json=true -> nonce/ssecurity + * 3. GET {location}&clientSign=... -> account-level serviceToken + * 4. GET account /pass/serviceLogin?sid=mimopc&callback=&_json=true + * 5. GET {api}/api/sts?...&ticket... -> Set-Cookie: serviceToken (mimopc scope) + */ + +const API_BASE = "https://mimo-server-cn.xiaomimimo.com"; +const ACCOUNT_HOST = "account.xiaomi.com"; +const API_UA = + "miNative PC/Normal Windows_NT/10.0.19045 SDKV/1.0.0 DEVT/PC DEVS/Windows APP/miaccount_desktop APPV/0.1.0"; +const SSO_UA = "MiClaw/1.0"; +const COOKIE_TTL_MS = 30 * 60 * 1000; + +// Per-account session caches (keyed by passToken hash) so multiple Xiaomi +// accounts / connections can rotate without clobbering each other. +const _cache = new Map(); // key -> { cookie, at } +const _inflight = new Map(); // key -> Promise + +function desktopCookiePath() { + const home = os.homedir(); + if (process.platform === "win32") { + return path.join(home, "AppData", "Roaming", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies"); + } + if (process.platform === "darwin") { + return path.join(home, "Library", "Application Support", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies"); + } + return path.join(home, ".config", "Xiaomi MiMo", "Partitions", "xiaomi-account", "Network", "Cookies"); +} + +/** + * Read the persisted Xiaomi account cookies from MiMo Desktop's Electron profile. + * The Chromium cookie DB is held with an exclusive lock while Desktop runs, so we + * copy it first and bail (return null) if that fails. + * @returns {Promise|null>} + */ +async function readDesktopAccountCookies() { + const src = desktopCookiePath(); + if (!fs.existsSync(src)) return null; + const tmp = path.join(os.tmpdir(), `9r-mimo-cookies-${process.pid}-${crypto.randomBytes(4).toString("hex")}.db`); + try { + fs.copyFileSync(src, tmp); + } catch { + return null; // locked by a running Desktop + } + try { + const { DatabaseSync } = await import("node:sqlite"); + const db = new DatabaseSync(tmp, { readOnly: true }); + const rows = db.prepare("SELECT name, value FROM cookies WHERE host_key = ?").all("." + ACCOUNT_HOST); + db.close(); + const jar = Object.fromEntries(rows.map((r) => [r.name, r.value])); + return jar.passToken ? jar : null; + } catch { + return null; + } finally { + try { + fs.unlinkSync(tmp); + } catch { + /* ignore */ + } + } +} + +/** + * Read just the passToken + identity cookies from Desktop's profile. + * Exported so the connect flow can persist a per-account passToken into the + * connection's providerSpecificData — this is what enables multi-account rotation. + * @returns {Promise<{passToken:string, userId:string|null, cUserId:string|null}|null>} + */ +export async function readDesktopPassToken() { + try { + const jar = await readDesktopAccountCookies(); + if (!jar?.passToken) return null; + return { passToken: jar.passToken, userId: jar.userId || null, cUserId: jar.cUserId || null }; + } catch { + return null; + } +} + +function signatureClientSign(nonce, ssecurity) { + const input = `nonce=${nonce}` + (ssecurity && ssecurity.trim() ? `&${ssecurity}` : ""); + return encodeURIComponent(crypto.createHash("sha1").update(input).digest("base64")); +} + +function absorbSetCookie(jar, res) { + for (const c of res.headers.getSetCookie?.() || []) { + const m = /^([^=]+)=([^;]*)/.exec(c.trim()); + if (m && m[2]) jar[m[1]] = m[2]; + } +} + +function cookieHeader(jar) { + return Object.entries(jar) + .filter(([, v]) => v) + .map(([k, v]) => `${k}=${v}`) + .join("; "); +} + +/** + * Exchange a passToken for a mimo-server service session cookie. + * @returns {Promise} Cookie header value, or null on failure. + */ +async function acquireServiceCookie(passJar, proxyOptions) { + const jar = { ...passJar }; + const ck = () => cookieHeader(jar); + + // 1. Unauthenticated API call -> 302 carrying the sts callback (sid=mimopc) + const r1 = await proxyAwareFetch( + `${API_BASE}/api/user/xiaomi/me`, + { redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } }, + proxyOptions, + ); + const redirect = r1.headers.get("location"); + if (!redirect) return null; + const stsCallback = new URL(redirect).searchParams.get("callback"); + if (!stsCallback) return null; + + // 2. passportapi SSO phase 1 -> nonce + ssecurity + const sso1 = await proxyAwareFetch( + `https://${ACCOUNT_HOST}/pass/serviceLogin?sid=passportapi&_json=true`, + { headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } }, + proxyOptions, + ); + const j1 = JSON.parse((await sso1.text()).replace(/^&&&START&&&/, "")); + const nonce = j1.nonce || (j1.location ? new URL(j1.location).searchParams.get("nonce") : null); + if (!nonce || !j1.location) return null; + + // 3. passportapi SSO phase 2 -> account-level serviceToken + const sso2 = await proxyAwareFetch( + `${j1.location}&clientSign=${signatureClientSign(nonce, j1.ssecurity)}`, + { redirect: "manual", headers: { Cookie: ck(), "User-Agent": SSO_UA } }, + proxyOptions, + ); + absorbSetCookie(jar, sso2); + + // 4. mimopc SSO -> sts callback carrying a ticket + const sso3 = await proxyAwareFetch( + `https://${ACCOUNT_HOST}/pass/serviceLogin?sid=mimopc&callback=${encodeURIComponent(stsCallback)}&_json=true`, + { headers: { Cookie: ck(), "User-Agent": SSO_UA, Accept: "application/json" } }, + proxyOptions, + ); + const j3 = JSON.parse((await sso3.text()).replace(/^&&&START&&&/, "")); + absorbSetCookie(jar, sso3); + if (!j3?.location || !/\/api\/sts/.test(j3.location)) return null; + + // 5. sts callback -> Set-Cookie: serviceToken (mimopc scope) + const sts = await proxyAwareFetch( + j3.location, + { redirect: "manual", headers: { "User-Agent": API_UA, Cookie: ck() } }, + proxyOptions, + ); + absorbSetCookie(jar, sts); + + const needed = ["serviceToken", "mimopc_ph", "mimopc_slh", "userId"]; + if (!jar.serviceToken) return null; + const out = {}; + for (const k of needed) if (jar[k]) out[k] = jar[k]; + return cookieHeader(out); +} + +/** + * Get (and cache) the mimo-server account cookie. + * @param {object|null} providerSpecificData - may carry `mimoPassToken` override + */ +async function getServiceCookie(providerSpecificData, proxyOptions) { + const passJar = providerSpecificData?.mimoPassToken + ? { passToken: providerSpecificData.mimoPassToken, userId: providerSpecificData.mimoUserId, cUserId: providerSpecificData.mimoCUserId } + : await readDesktopAccountCookies(); + if (!passJar) return { cookie: null, reason: "no-pass-token" }; + + // One cached session per passToken — accounts/connections rotate independently. + const key = crypto.createHash("sha256").update(passJar.passToken).digest("hex"); + + const cached = _cache.get(key); + if (cached && Date.now() - cached.at < COOKIE_TTL_MS) { + return { cookie: cached.cookie }; + } + + // De-dupe concurrent handshakes for the same account: a burst of requests must + // not each run the full 5-step SSO chain. + const inflight = _inflight.get(key); + if (inflight) { + const cookie = await inflight; + return cookie ? { cookie } : { cookie: null, reason: "sso-failed" }; + } + + const promise = (async () => { + try { + return await acquireServiceCookie(passJar, proxyOptions); + } catch { + return null; // network/parse failure — callers degrade, never throw + } finally { + _inflight.delete(key); + } + })(); + _inflight.set(key, promise); + + const cookie = await promise; + if (!cookie) return { cookie: null, reason: "sso-failed" }; + _cache.set(key, { cookie, at: Date.now() }); + return { cookie }; +} + +/** Drop cached sessions so the next call re-runs the handshake (e.g. after a 401). */ +export function invalidateMimoAccountCookieCache() { + _cache.clear(); +} + +/** mimo-server account API base + the User-Agent its backend expects. */ +export const MIMO_API_BASE = API_BASE; +export const MIMO_API_UA = API_UA; + +/** + * Resolve the mimo-server account-session cookie, for upstream /api/route/* calls. + * @returns {Promise} Cookie header value, or null when unavailable. + */ +export async function getMimoAccountCookie(providerSpecificData = null, proxyOptions = null) { + try { + const { cookie } = await getServiceCookie(providerSpecificData, proxyOptions); + return cookie; + } catch { + return null; + } +} + +/** + * Fetch the weekly quota from the account service. + * @returns {Promise<{percent?:number, resetDate?:string, resetAt?:number, error?:string}>} + */ +export async function getMimoAccountUsage(providerSpecificData = null, proxyOptions = null) { + const { cookie, reason } = await getServiceCookie(providerSpecificData, proxyOptions); + if (!cookie) { + return { error: reason === "no-pass-token" ? "no-session" : "session-failed" }; + } + try { + const res = await proxyAwareFetch( + `${API_BASE}/api/user/usage`, + { headers: { "User-Agent": API_UA, Cookie: cookie, Accept: "application/json" }, signal: AbortSignal.timeout(10000) }, + proxyOptions, + ); + if (!res.ok) return { error: `http-${res.status}` }; + const data = await res.json().catch(() => null); + if (!data || data.code !== 0 || !data.data) return { error: "bad-response" }; + return { percent: data.data.percent, resetDate: data.data.resetDate, resetAt: data.data.resetAt }; + } catch (e) { + return { error: e.message }; + } +} diff --git a/open-sse/shared/qoder/attachments.js b/open-sse/shared/qoder/attachments.js new file mode 100644 index 00000000..d2024529 --- /dev/null +++ b/open-sse/shared/qoder/attachments.js @@ -0,0 +1,341 @@ +/** + * Native qodercli does NOT stuff image/PDF bytes into agent_chat_generation. + * It PUTs them to /algo/api/v2/image/upload (COSY-signed multipart) and then + * sends the returned OSS URL. Agents like Claude Code send OpenAI/Claude + * data-URIs instead, which 9router previously forwarded verbatim — 10MB + * images become 30MB+ JSON and upstream 413s even though the model window + * is ~200k tokens. + * + * This module: + * 1. Uploads inlined images to Qoder's file API (cached by sha256). + * 2. Replaces huge non-image file blocks with a short stub. + * 3. Caps leftover data-URIs so the chat JSON stays small. + */ + +import { createHash } from "crypto"; +import { v4 as uuidv4 } from "uuid"; + +import { proxyAwareFetch } from "../../utils/proxyFetch.js"; +import { parseDataUri } from "../../translator/concerns/image.js"; +import { OPENAI_BLOCK, CLAUDE_BLOCK } from "../../translator/schema/blocks.js"; +import { MAX_IMAGE_BYTES } from "../../config/mediaConfig.js"; +import { buildCosyHeaders } from "./cosy.js"; +import { + QODER_IMAGE_UPLOAD_SIG_PATH, + QODER_INLINE_FALLBACK_MAX_BYTES, + QODER_MAX_PAYLOAD_BYTES, + qoderInferenceBase, +} from "./constants.js"; + +const IMAGE_MIME_RE = /^image\//i; +const DATA_URI_RE = /data:[^;]+;base64,[A-Za-z0-9+/=\s]+/g; + +function mimeExt(mime) { + const m = String(mime || "").toLowerCase(); + if (m.includes("png")) return "png"; + if (m.includes("jpeg") || m.includes("jpg")) return "jpg"; + if (m.includes("gif")) return "gif"; + if (m.includes("webp")) return "webp"; + if (m.includes("bmp")) return "bmp"; + if (m.includes("pdf")) return "pdf"; + return "bin"; +} + +function decodedBytes(b64) { + if (typeof b64 !== "string" || !b64) return 0; + const compact = b64.replace(/\s/g, ""); + return Math.floor(compact.length * 3 / 4); +} + +function stubText({ name, mime, bytes, reason }) { + const label = name || mime || "attachment"; + const size = bytes ? `, ${bytes} bytes` : ""; + return `[file omitted: ${label}${size} — ${reason}]`; +} + +export function buildMultipartFile(buffer, { fieldName = "file", fileName, mediaType } = {}) { + const boundary = `----9routerQoder${Date.now().toString(16)}${Math.random().toString(16).slice(2)}`; + const filename = fileName || `upload.${mimeExt(mediaType)}`; + const head = Buffer.from( + `--${boundary}\r\nContent-Disposition: form-data; name="${fieldName}"; filename="${filename}"\r\nContent-Type: ${mediaType || "application/octet-stream"}\r\n\r\n`, + ); + const tail = Buffer.from(`\r\n--${boundary}--\r\n`); + const body = Buffer.concat([head, buffer, tail]); + return { boundary, body }; +} + +function extractUrlFromUploadResponse(json) { + if (!json || typeof json !== "object") return null; + const result = json.result && typeof json.result === "object" ? json.result : json; + const arrays = [result.imageUrls, result.image_urls, json.imageUrls, json.image_urls]; + for (const arr of arrays) { + if (Array.isArray(arr) && typeof arr[0] === "string" && arr[0]) return arr[0]; + } + const keys = ["imageUrl", "image_url", "url", "ossUrl", "oss_url", "originalUrl", "originUrl", "link", "image"]; + for (const key of keys) { + const v = result[key] ?? json[key]; + if (typeof v === "string" && v) return v; + } + if (typeof json.body === "string") { + try { return extractUrlFromUploadResponse(JSON.parse(json.body)); } catch { /* ignore */ } + } + return null; +} + +async function defaultUploadImage({ buffer, mediaType, credentials, proxyOptions, signal }) { + const requestId = uuidv4(); + const url = `${qoderInferenceBase(credentials)}${`/algo${QODER_IMAGE_UPLOAD_SIG_PATH}`}?request_id=${requestId}`; + const { boundary, body } = buildMultipartFile(buffer, { + fileName: `image.${mimeExt(mediaType)}`, + mediaType: mediaType || "application/octet-stream", + }); + const psd = credentials?.providerSpecificData || {}; + const cosyHeaders = buildCosyHeaders(body, url, { + userId: psd.userId, + authToken: credentials.accessToken, + name: credentials.displayName || "", + email: credentials.email || "", + machineId: psd.machineId || "", + }); + const headers = { + ...cosyHeaders, + Accept: "application/json", + "Content-Type": `multipart/form-data; boundary=${boundary}`, + "Content-Length": String(body.length), + "AI-CLIENT-TIMESTAMP": String(Math.floor(Date.now() / 1000)), + "Accept-Encoding": "identity", + }; + const res = await proxyAwareFetch( + url, + { method: "PUT", headers, body, signal }, + proxyOptions, + ); + if (!res.ok) { + const text = await res.text().catch(() => ""); + throw new Error(`HTTP ${res.status}${text ? `: ${text.slice(0, 180)}` : ""}`); + } + const json = await res.json().catch(() => null); + const uploaded = extractUrlFromUploadResponse(json); + if (!uploaded) throw new Error("upload response missing url"); + return uploaded; +} + +async function uploadImageData({ base64, mediaType, credentials, proxyOptions, signal, log, uploadFn, cache }) { + const compact = String(base64 || "").replace(/\s/g, ""); + if (!compact) return null; + const bytes = decodedBytes(compact); + if (bytes > MAX_IMAGE_BYTES) { + log?.warn?.("QODER", `image ${bytes} bytes exceeds upload cap, stubbing`); + return { stub: true, bytes, mime: mediaType }; + } + let buffer; + try { + buffer = Buffer.from(compact, "base64"); + } catch { + return { stub: true, bytes, mime: mediaType }; + } + const digest = createHash("sha256").update(buffer).digest("hex"); + if (cache?.has(digest)) return { url: cache.get(digest), bytes, mime: mediaType }; + + const doUpload = uploadFn || defaultUploadImage; + try { + const url = await doUpload({ buffer, mediaType, credentials, proxyOptions, signal }); + if (typeof url === "string" && url) { + cache?.set(digest, url); + return { url, bytes, mime: mediaType }; + } + } catch (err) { + log?.warn?.("QODER", `image upload failed (${err.message}); ${bytes <= QODER_INLINE_FALLBACK_MAX_BYTES ? "keeping inline" : "stubbing"}`); + } + if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return { keep: true, bytes, mime: mediaType }; + return { stub: true, bytes, mime: mediaType }; +} + +function imageUrlBlock(url) { + return { type: OPENAI_BLOCK.IMAGE_URL, image_url: { url } }; +} + +async function rewriteBlock(block, ctx) { + if (!block || typeof block !== "object") return block; + + if (block.type === OPENAI_BLOCK.IMAGE_URL) { + const raw = typeof block.image_url === "string" ? block.image_url : block.image_url?.url; + if (typeof raw !== "string" || !raw) return null; + if (raw.startsWith("http://") || raw.startsWith("https://")) return imageUrlBlock(raw); + const parsed = parseDataUri(raw); + if (!parsed) return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "attachment", reason: "unreadable data URI" }) }; + if (!IMAGE_MIME_RE.test(parsed.mimeType)) { + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "file", mime: parsed.mimeType, bytes: decodedBytes(parsed.base64), reason: "non-image bytes are not inlined into Qoder context" }) }; + } + const up = await uploadImageData({ ...ctx, base64: parsed.base64, mediaType: parsed.mimeType }); + if (up?.url) return imageUrlBlock(up.url); + if (up?.keep) return imageUrlBlock(raw); + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", mime: parsed.mimeType, bytes: up?.bytes, reason: "upload failed; not inlined" }) }; + } + + if (block.type === OPENAI_BLOCK.IMAGE || block.type === CLAUDE_BLOCK.IMAGE) { + const src = block.source || {}; + if (src.type === "url" && typeof src.url === "string") return imageUrlBlock(src.url); + if (src.type === "base64" && src.data) { + const mime = src.media_type || "image/png"; + const up = await uploadImageData({ ...ctx, base64: src.data, mediaType: mime }); + if (up?.url) return imageUrlBlock(up.url); + if (up?.keep) return imageUrlBlock(`data:${mime};base64,${src.data}`); + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", mime, bytes: up?.bytes, reason: "upload failed; not inlined" }) }; + } + } + + if (block.type === OPENAI_BLOCK.FILE && block.file) { + const file = block.file; + const name = file.filename || file.name || "file"; + const dataUri = typeof file.file_data === "string" ? file.file_data : null; + const parsed = dataUri ? parseDataUri(dataUri) : null; + const b64 = parsed?.base64 || (typeof file.file_data === "string" && !file.file_data.startsWith("data:") ? file.file_data : null); + const mime = parsed?.mimeType || file.format || "application/octet-stream"; + if (b64 && IMAGE_MIME_RE.test(mime)) { + const up = await uploadImageData({ ...ctx, base64: b64, mediaType: mime }); + if (up?.url) return imageUrlBlock(up.url); + } + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name, mime, bytes: decodedBytes(b64 || ""), reason: "Qoder reads documents via its file API, not inlined bytes" }) }; + } + + if (block.type === CLAUDE_BLOCK.DOCUMENT && block.source) { + const src = block.source; + const name = block.title || "document"; + if (src.type === "base64" && src.data) { + const mime = src.media_type || "application/pdf"; + if (IMAGE_MIME_RE.test(mime)) { + const up = await uploadImageData({ ...ctx, base64: src.data, mediaType: mime }); + if (up?.url) return imageUrlBlock(up.url); + } + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name, mime, bytes: decodedBytes(src.data), reason: "Qoder reads documents via its file API, not inlined bytes" }) }; + } + } + + if (typeof block.text === "string" && block.text.includes("data:") && block.text.length > 8192) { + const next = block.text.replace(DATA_URI_RE, (m) => { + const parsed = parseDataUri(m.trim()); + const bytes = parsed ? decodedBytes(parsed.base64) : m.length; + if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return m; + return stubText({ mime: parsed?.mimeType, bytes, reason: "inlined data URI stripped from Qoder context" }); + }); + return { ...block, text: next }; + } + + return block; +} + +async function rewriteContent(content, ctx) { + if (typeof content === "string") { + if (content.includes("data:") && content.length > 8192) { + return content.replace(DATA_URI_RE, (m) => { + const parsed = parseDataUri(m.trim()); + const bytes = parsed ? decodedBytes(parsed.base64) : m.length; + if (bytes <= QODER_INLINE_FALLBACK_MAX_BYTES) return m; + return stubText({ mime: parsed?.mimeType, bytes, reason: "inlined data URI stripped from Qoder context" }); + }); + } + return content; + } + if (!Array.isArray(content)) return content; + const out = []; + for (const block of content) { + const next = await rewriteBlock(block, ctx); + if (next == null) continue; + out.push(next); + } + return out.length ? out : ""; +} + +function payloadBytes(messages) { + try { + return Buffer.byteLength(JSON.stringify(messages), "utf8"); + } catch { + return 0; + } +} + +function stripRemainingDataUris(messages) { + for (const msg of messages || []) { + if (typeof msg?.content === "string" && msg.content.includes("data:")) { + msg.content = msg.content.replace(DATA_URI_RE, (m) => + stubText({ bytes: m.length, reason: "payload over Qoder size budget" }), + ); + } else if (Array.isArray(msg?.content)) { + msg.content = msg.content.map((block) => { + if (block?.type === OPENAI_BLOCK.IMAGE_URL) { + const raw = typeof block.image_url === "string" ? block.image_url : block.image_url?.url; + if (typeof raw === "string" && raw.startsWith("data:")) { + return { type: OPENAI_BLOCK.TEXT, text: stubText({ name: "image", reason: "payload over Qoder size budget" }) }; + } + } + if (typeof block?.text === "string" && block.text.includes("data:")) { + return { ...block, text: block.text.replace(DATA_URI_RE, (m) => + stubText({ bytes: m.length, reason: "payload over Qoder size budget" }), + ) }; + } + return block; + }); + } + } +} + +/** + * Rewrite OpenAI-shaped messages in place: upload images, stub huge files. + * @returns {Promise<{imageUrls: string[], uploaded: number, stubbed: number}>} + */ +export async function rewriteQoderMessageAttachments(messages, { + credentials, + log, + proxyOptions = null, + signal = null, + uploadFn = null, +} = {}) { + const stats = { imageUrls: [], uploaded: 0, stubbed: 0 }; + if (!Array.isArray(messages) || messages.length === 0) return stats; + + const ctx = { credentials, log, proxyOptions, signal, uploadFn, cache: new Map() }; + + for (const msg of messages) { + if (!msg || typeof msg !== "object") continue; + if (Array.isArray(msg.images)) { + // Ollama-style sidecar; fold into content so normalizeMessages can see them. + const extras = msg.images.map((url) => imageUrlBlock(String(url))); + msg.content = Array.isArray(msg.content) + ? [...msg.content, ...extras] + : [{ type: OPENAI_BLOCK.TEXT, text: typeof msg.content === "string" ? msg.content : "" }, ...extras]; + delete msg.images; + } + msg.content = await rewriteContent(msg.content, ctx); + } + + // Collect surviving http(s) image URLs for callers that want image_urls. + for (const msg of messages) { + if (!Array.isArray(msg?.content)) continue; + for (const block of msg.content) { + const url = block?.type === OPENAI_BLOCK.IMAGE_URL + ? (typeof block.image_url === "string" ? block.image_url : block.image_url?.url) + : null; + if (typeof url === "string" && /^https?:\/\//i.test(url)) stats.imageUrls.push(url); + if (block?.type === OPENAI_BLOCK.TEXT && typeof block.text === "string" && block.text.startsWith("[file omitted:")) stats.stubbed += 1; + } + } + stats.uploaded = stats.imageUrls.length; + + if (payloadBytes(messages) > QODER_MAX_PAYLOAD_BYTES) { + log?.warn?.("QODER", `request still ${payloadBytes(messages)} bytes after rewrite; stripping leftover data URIs`); + stripRemainingDataUris(messages); + } + + return stats; +} + +/** Test helper kept for callers; upload memo is now per-request. */ +export function clearQoderUploadCache() {} + +export const __test__ = { + extractUrlFromUploadResponse, + decodedBytes, + stubText, + payloadBytes, +}; diff --git a/open-sse/shared/qoder/constants.js b/open-sse/shared/qoder/constants.js index e2635f40..849f67ed 100644 --- a/open-sse/shared/qoder/constants.js +++ b/open-sse/shared/qoder/constants.js @@ -33,6 +33,39 @@ export const QODER_CHAT_SIG_PATH = "/api/v2/service/pro/sse/agent_chat_generatio export const QODER_CHAT_URL = `${QODER_CHAT_BASE}/algo${QODER_CHAT_SIG_PATH}?FetchKeys=llm_model_result&AgentId=agent_common`; export const QODER_CHAT_URL_ENCODED = `${QODER_CHAT_URL}&Encode=1`; export const QODER_MODEL_LIST_URL = `${QODER_CHAT_BASE}/algo/api/v2/model/list`; +// Official qodercli uploads images here (COSY-signed PUT multipart, field "file") +// instead of inlining base64 into agent_chat_generation. +export const QODER_IMAGE_UPLOAD_SIG_PATH = "/api/v2/image/upload"; + +// Drop remaining inlined binaries if the Qoder JSON body would still exceed this. +// 30MB+ payloads are what blow past Claude-Code's ~200k context on the wire. +export const QODER_MAX_PAYLOAD_BYTES = 6 * 1024 * 1024; +// If OSS upload fails, keep tiny data-URIs; anything larger is stubbed. +export const QODER_INLINE_FALLBACK_MAX_BYTES = 512 * 1024; + +// Context-window tier selection (see shared/qoder/contextTier.js). The IDE exposes the +// model's context_config tiers (200K/400K/1M); we auto-escalate when the estimated prompt +// (+ headroom, tokenizer variance) no longer fits the current max_input_tokens. +export const QODER_CONTEXT_TIER_HEADROOM = 0.15; +export const QODER_CONTEXT_TIER_ENV = "QODER_CONTEXT_TIER"; +export const QODER_CONTEXT_TIER_MODES = Object.freeze({ AUTO: "auto", MAX: "max", DEFAULT: "default" }); + +/** + * Job-token (jt-...) traffic must hit api2.qoder.sh — api3 rejects jt- with + * "Login expired" (403). Device tokens (dt-...) stay on api3. PATs (pt-...) + * are exchanged for jt- before this is consulted. + */ +export function qoderInferenceBase(credentials) { + const raw = credentials?.apiKey || credentials?.accessToken; + if ( + typeof raw === "string" && + !raw.startsWith("pt-") && + (raw.startsWith("jt-") || (credentials?.accessToken || "").startsWith("jt-")) + ) { + return QODER_CHAT_BASE_ALT; + } + return QODER_CHAT_BASE; +} // COSY header constants. These are not arbitrary — the upstream signature // validation matches them against the values used at signing time. @@ -54,10 +87,13 @@ export const QODER_MODEL_MAP = { lite: "lite", // Frontier models qmodel: "qmodel", + qfmodel: "qfmodel", qmodel_latest: "qmodel_latest", + qmodel_38max: "qmodel_38max", dmodel: "dmodel", dfmodel: "dfmodel", - gm51model: "gm51model", + gmodel: "gmodel", + gfmodel: "gfmodel", kmodel: "kmodel", mmodel: "mmodel", }; diff --git a/open-sse/shared/qoder/contextTier.js b/open-sse/shared/qoder/contextTier.js new file mode 100644 index 00000000..523c2fe5 --- /dev/null +++ b/open-sse/shared/qoder/contextTier.js @@ -0,0 +1,160 @@ +/** + * Qoder context-window tiers. + * + * Each Qoder model_config ships a `context_config` list (e.g. 200K / 400K / 1M for + * qmodel_38max) while `max_input_tokens` only carries the tier the IDE currently has + * selected (~180K by default). The Qoder IDE lets the user switch tiers from the model + * picker; a qodercli-style client (which is what 9router impersonates) has no picker, + * so a long Claude-Code / Codex session that grew past the default tier is rejected + * upstream even though the model itself supports 1M. + * + * This module emulates the IDE: estimate the prompt size, pick the smallest advertised + * tier that fits (never below the model's current default), and mirror the choice into + * the same three places the IDE writes: + * parameters.context_length + * chat_context.extra.ideModelConfigOverride.max_input_tokens + * model_config.max_input_tokens + * + * Override with QODER_CONTEXT_TIER = auto (default) | max | default | . + * Pure functions, no I/O — the executor wires them into buildQoderRequestBody. + */ + +import { QODER_CONTEXT_TIER_HEADROOM, QODER_CONTEXT_TIER_MODES } from "./constants.js"; + +const UNIT = { K: 1_000, M: 1_000_000 }; + +/** "200K" | "1M" | "204800" | 204800 → integer token count (0 when unparseable). */ +export function parseTierTokenCount(value) { + if (typeof value === "number") return Number.isFinite(value) && value > 0 ? Math.floor(value) : 0; + if (typeof value !== "string") return 0; + const m = value.trim().toUpperCase().match(/^(\d+(?:\.\d+)?)\s*([KM])?$/); + if (!m) return 0; + const n = Number(m[1]) * (UNIT[m[2]] || 1); + return Number.isFinite(n) && n > 0 ? Math.floor(n) : 0; +} + +function tierName(entry, tokenCount) { + const raw = entry.name ?? entry.label ?? entry.display_name ?? entry.displayName ?? entry.key ?? entry.id; + if (typeof raw === "string" && raw.trim()) return raw.trim(); + if (tokenCount >= UNIT.M && tokenCount % UNIT.M === 0) return `${tokenCount / UNIT.M}M`; + if (tokenCount >= UNIT.K && tokenCount % UNIT.K === 0) return `${tokenCount / UNIT.K}K`; + return String(tokenCount); +} + +/** + * Normalize a model_config into sorted tiers: [{ name, tokenCount, isDefault }] ascending. + * Accepts snake_case and camelCase shapes; returns [] when the model has no tiers. + */ +export function getQoderContextTiers(modelConfig) { + const list = modelConfig?.context_config ?? modelConfig?.contextConfig; + if (!Array.isArray(list)) return []; + const byCount = new Map(); + for (const entry of list) { + if (!entry || typeof entry !== "object") continue; + const tokenCount = parseTierTokenCount( + entry.tokenCount ?? entry.token_count ?? entry.max_input_tokens ?? entry.maxInputTokens ?? entry.contextLength ?? entry.context_length, + ); + if (!tokenCount) continue; + const isDefault = entry.isDefault === true || entry.is_default === true || entry.default === true; + const prev = byCount.get(tokenCount); + byCount.set(tokenCount, { + name: tierName(entry, tokenCount), + tokenCount, + isDefault: (prev?.isDefault || false) || isDefault, + }); + } + return [...byCount.values()].sort((a, b) => a.tokenCount - b.tokenCount); +} + +const CJK_RE = /[\u1100-\u11ff\u2e80-\u9fff\uac00-\ud7af\uf900-\ufaff\uff00-\uffef]/g; + +/** + * Rough prompt-size estimate in tokens. CJK characters count ~1 token each, everything + * else ~4 chars/token — the plain chars/4 rule underestimates Chinese/Japanese by up to + * 4x, which is exactly when a tier decision matters. + */ +export function estimateQoderPromptTokens({ system, messages, tools } = {}) { + let text = ""; + try { + text = JSON.stringify({ system: system || "", messages: messages || [], tools: tools || [] }) || ""; + } catch { + return 0; + } + const cjk = (text.match(CJK_RE) || []).length; + return Math.ceil(cjk + (text.length - cjk) / 4); +} + +function normalizeMode(preference) { + const p = String(preference ?? "").trim(); + return p ? p : QODER_CONTEXT_TIER_MODES.AUTO; +} + +function findNamedTier(tiers, name) { + const wanted = name.replace(/\s+/g, "").toUpperCase(); + const asCount = parseTierTokenCount(wanted); + return tiers.find((t) => t.name.replace(/\s+/g, "").toUpperCase() === wanted || (asCount && t.tokenCount === asCount)) || null; +} + +/** + * Decide which tier a request should run under. + * + * @param {object} modelConfig raw Qoder model_config (has context_config + max_input_tokens) + * @param {{system?: string, messages?: any[], tools?: any[]}} prompt what will be sent + * @param {{preference?: string, headroom?: number}} [options] + * @returns {{ tier: {name, tokenCount, isDefault}, estimatedTokens: number, reason: string } | null} + * null → leave the payload exactly as before (no tiers, or the default already fits). + */ +export function resolveQoderContextTier(modelConfig, prompt, options = {}) { + const tiers = getQoderContextTiers(modelConfig); + if (!tiers.length) return null; + + const mode = normalizeMode(options.preference); + const largest = tiers[tiers.length - 1]; + const defaultTier = tiers.find((t) => t.isDefault) || tiers[0]; + const estimatedTokens = estimateQoderPromptTokens(prompt); + const headroom = typeof options.headroom === "number" ? options.headroom : QODER_CONTEXT_TIER_HEADROOM; + const need = Math.ceil(estimatedTokens * (1 + headroom)); + + if (mode.toLowerCase() === QODER_CONTEXT_TIER_MODES.MAX) { + return { tier: largest, estimatedTokens, reason: "forced:max" }; + } + if (mode.toLowerCase() === QODER_CONTEXT_TIER_MODES.DEFAULT) { + return { tier: defaultTier, estimatedTokens, reason: "forced:default" }; + } + if (mode.toLowerCase() !== QODER_CONTEXT_TIER_MODES.AUTO) { + const named = findNamedTier(tiers, mode); + if (named) return { tier: named, estimatedTokens, reason: `forced:${named.name}` }; + // Unknown tier name → fall through to auto rather than silently breaking requests. + } + + // auto: keep the upstream default (current behaviour) while the prompt fits in it. + const currentMax = parseTierTokenCount(modelConfig?.max_input_tokens ?? modelConfig?.maxInputTokens); + const currentLimit = currentMax || defaultTier.tokenCount; + if (need <= currentLimit) return null; + + const fits = tiers.find((t) => t.tokenCount >= need && t.tokenCount > currentLimit); + const tier = fits || largest; + if (tier.tokenCount <= currentLimit) return null; // nothing bigger to escalate to + return { tier, estimatedTokens, reason: fits ? "auto:fits" : "auto:largest" }; +} + +/** + * Write the chosen tier into a Qoder chat payload (mutates + returns it). + * Mirrors the IDE: parameters.context_length, ideModelConfigOverride, model_config. + */ +export function applyQoderContextTier(payload, tier) { + if (!payload || !tier?.tokenCount) return payload; + payload.parameters = { ...(payload.parameters || {}), context_length: tier.tokenCount }; + payload.chat_context = payload.chat_context || {}; + payload.chat_context.extra = { + ...(payload.chat_context.extra || {}), + ideModelConfigOverride: { + ...(payload.chat_context.extra?.ideModelConfigOverride || {}), + max_input_tokens: tier.tokenCount, + }, + }; + if (payload.model_config && typeof payload.model_config === "object") { + payload.model_config = { ...payload.model_config, max_input_tokens: tier.tokenCount }; + } + return payload; +} diff --git a/open-sse/shared/qoder/sse.js b/open-sse/shared/qoder/sse.js new file mode 100644 index 00000000..ad30785e --- /dev/null +++ b/open-sse/shared/qoder/sse.js @@ -0,0 +1,208 @@ +/** + * Qoder SSE is OpenAI-shaped inside `{statusCodeValue, body}` envelopes, but + * usage arrives on a later `choices: []` frame — after finish_reason, which + * itself often lives on `delta.finish_reason` rather than the choice. + * + * Downstream (Claude translator, OpenAI clients, Claude Code) look for usage + * on the finish chunk or drop `choices: []` entirely. 9router's own dashboard + * still sees tokens because extractUsage runs on every forwarded frame. + * + * Coalesce: hold empty finish + usage-only frames, then emit one OpenAI + * include_usage-style chunk: `{choices:[{delta:{}, finish_reason}], usage}`. + */ + +function num(v) { + const n = Number(v); + return Number.isFinite(n) ? n : null; +} + +/** + * Normalize Qoder/OpenAI usage into the shape stream.js + Claude translation + * already understand (prompt_tokens + prompt_tokens_details.cached_tokens). + */ +export function canonicalizeQoderUsage(usage) { + if (!usage || typeof usage !== "object" || Array.isArray(usage)) return null; + + const prompt = num(usage.prompt_tokens ?? usage.input_tokens); + const completion = num(usage.completion_tokens ?? usage.output_tokens); + if (prompt == null && completion == null) return null; + + const details = (usage.prompt_tokens_details && typeof usage.prompt_tokens_details === "object") + ? { ...usage.prompt_tokens_details } + : {}; + const cached = num( + details.cached_tokens ?? + usage.cached_tokens ?? + usage.prompt_cache_hit_tokens ?? + usage.cache_read_input_tokens, + ); + const cacheCreation = num( + details.cache_creation_tokens ?? + usage.cache_creation_input_tokens, + ); + + const promptTokens = prompt || 0; + const completionTokens = completion || 0; + const out = { + prompt_tokens: promptTokens, + completion_tokens: completionTokens, + total_tokens: num(usage.total_tokens) ?? (promptTokens + completionTokens), + }; + + if (cached != null) { + out.cached_tokens = cached; + details.cached_tokens = cached; + } + if (cacheCreation != null) { + details.cache_creation_tokens = cacheCreation; + } + if (Object.keys(details).length) out.prompt_tokens_details = details; + + if (usage.completion_tokens_details && typeof usage.completion_tokens_details === "object") { + out.completion_tokens_details = usage.completion_tokens_details; + } + const reasoning = num(usage.reasoning_tokens ?? usage.completion_tokens_details?.reasoning_tokens); + if (reasoning != null) out.reasoning_tokens = reasoning; + + return out; +} + +function finishReasonOf(parsed) { + const choice = parsed?.choices?.[0]; + return choice?.finish_reason || choice?.delta?.finish_reason || parsed?.finish_reason || null; +} + +function hasValuableDelta(parsed) { + const delta = parsed?.choices?.[0]?.delta; + if (!delta || typeof delta !== "object") return false; + if (typeof delta.content === "string" && delta.content.length > 0) return true; + if (typeof delta.reasoning_content === "string" && delta.reasoning_content.length > 0) return true; + if (Array.isArray(delta.tool_calls) && delta.tool_calls.length > 0) return true; + if (delta.role) return true; + return false; +} + +function parseInner(inner) { + if (inner == null || inner === "") return { raw: false, parsed: null }; + if (inner === "[DONE]") return { done: true }; + if (typeof inner !== "string") { + if (typeof inner === "object") return { parsed: inner }; + return { raw: true, text: String(inner) }; + } + try { + return { parsed: JSON.parse(inner) }; + } catch { + return { raw: true, text: inner }; + } +} + +/** + * @param {object} opts + * @param {string} opts.model + * @param {TextEncoder} opts.encoder + * @param {string} opts.sseDone "data: [DONE]\\n\\n" + */ +export function createQoderSseCoalescer({ model, encoder, sseDone }) { + let pendingFinish = null; + let pendingUsage = null; + let lastMeta = { id: null, created: null, model }; + let doneEmitted = false; + let finishAlreadyForwarded = false; + + const emitJson = (controller, obj) => { + const sanitized = JSON.stringify(obj).replace(/\r?\n/g, ""); + controller.enqueue(encoder.encode(`data: ${sanitized}\n\n`)); + }; + + const emitRaw = (controller, text) => { + controller.enqueue(encoder.encode(`data: ${String(text).replace(/\r?\n/g, "")}\n\n`)); + }; + + const emitDone = (controller) => { + if (doneEmitted) return; + controller.enqueue(encoder.encode(sseDone)); + doneEmitted = true; + }; + + const emitTerminal = (controller) => { + if (!pendingFinish && !pendingUsage) return; + emitJson(controller, { + id: lastMeta.id || `qoder-${Date.now()}`, + object: "chat.completion.chunk", + created: lastMeta.created || Math.floor(Date.now() / 1000), + model: lastMeta.model || model, + choices: [{ index: 0, delta: {}, finish_reason: pendingFinish || "stop" }], + ...(pendingUsage ? { usage: pendingUsage } : {}), + }); + pendingFinish = null; + pendingUsage = null; + }; + + const flush = (controller) => { + if (doneEmitted) return; + if (pendingUsage || (pendingFinish && !finishAlreadyForwarded)) { + emitTerminal(controller); + } + emitDone(controller); + }; + + const handleInner = (inner, controller) => { + if (doneEmitted) return { terminal: true }; + + const parsedInner = parseInner(inner); + if (parsedInner.done) { + flush(controller); + return { terminal: true }; + } + if (parsedInner.raw) { + emitRaw(controller, parsedInner.text); + return {}; + } + const parsed = parsedInner.parsed; + if (!parsed || typeof parsed !== "object") return {}; + + if (typeof parsed.id === "string" && parsed.id) lastMeta.id = parsed.id; + if (typeof parsed.created === "number") lastMeta.created = parsed.created; + if (typeof parsed.model === "string" && parsed.model) lastMeta.model = parsed.model; + + const usage = canonicalizeQoderUsage(parsed.usage); + if (usage) pendingUsage = usage; + + const finish = finishReasonOf(parsed); + if (hasValuableDelta(parsed)) { + // Stream content as-is (preserves upstream JSON for tests/clients). + emitRaw(controller, typeof inner === "string" ? inner : JSON.stringify(parsed)); + if (finish) { + finishAlreadyForwarded = true; + // Keep finish around only if we still need a usage trailer. + pendingFinish = pendingUsage ? finish : null; + } + if (pendingFinish && pendingUsage) { + emitTerminal(controller); + emitDone(controller); + return { terminal: true }; + } + return {}; + } + + if (finish) pendingFinish = finish; + + // Empty finish and/or usage-only: emit as soon as we have both (Qoder + // order is finish then usage). Don't wait for the later [DONE]/keepalive. + if ((pendingFinish || finishAlreadyForwarded) && pendingUsage) { + if (!pendingFinish) pendingFinish = "stop"; + emitTerminal(controller); + emitDone(controller); + return { terminal: true }; + } + return {}; + }; + + return { + handleInner, + flush, + get doneEmitted() { + return doneEmitted; + }, + }; +} diff --git a/open-sse/shared/zedAuth.js b/open-sse/shared/zedAuth.js index e3d8371a..4bfc33bc 100644 --- a/open-sse/shared/zedAuth.js +++ b/open-sse/shared/zedAuth.js @@ -112,7 +112,14 @@ export function parseZedCallbackPayload(input) { url = new URL(raw); } catch { try { - url = new URL(`http://127.0.0.1/?${raw.replace(/^\?/, "")}`); + // Accept pathname+query (what the local proxy forwards, e.g. + // "/?user_id=..&access_token=.." or "/callback?.."), a bare query, + // or a lone query string. Only the query part is parsed — a leading + // path must never become part of the first parameter name. + const query = raw.includes("?") + ? raw.slice(raw.indexOf("?") + 1) + : raw.replace(/^\?/, ""); + url = new URL(`http://127.0.0.1/?${query}`); } catch { throw new Error("Invalid Zed callback URL"); } @@ -134,6 +141,10 @@ export function parseZedCallbackPayload(input) { export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier) { const privateKey = decodeZedPrivateKeyVerifier(privateKeyVerifier); const encrypted = Buffer.from(String(encryptedAccessToken), "base64url"); + const fail = (oaepError) => { + const message = oaepError instanceof Error ? oaepError.message : String(oaepError); + throw new Error(`Failed to decrypt Zed access token: ${message}`); + }; try { return crypto .privateDecrypt( @@ -143,15 +154,21 @@ export function decryptZedAccessToken(encryptedAccessToken, privateKeyVerifier) .toString("utf8"); } catch (oaepError) { try { - return crypto + const text = crypto .privateDecrypt( { key: privateKey, padding: crypto.constants.RSA_PKCS1_PADDING }, encrypted, ) .toString("utf8"); - } catch { - const message = oaepError instanceof Error ? oaepError.message : String(oaepError); - throw new Error(`Failed to decrypt Zed access token: ${message}`); + // PKCS#1 v1.5 unpadding is not integrity-checked: a wrong-key decrypt + // can "succeed" with garbage bytes instead of throwing. Replacement + // characters prove the output is not the real UTF-8 token — fail loudly + // rather than storing garbage as a credential. + if (text.includes("�")) fail(oaepError); + return text; + } catch (err) { + if (err.message.startsWith("Failed to decrypt Zed access token")) throw err; + fail(oaepError); } } } @@ -280,6 +297,7 @@ export async function fetchZedLlmToken(credentials, options = {}) { body: JSON.stringify({ organization_id: organizationId }), signal: options.signal ?? undefined, }, + options.proxyOptions ?? null, ); const token = typeof data?.token === "string" ? data.token : data?.token?.[0] || data?.token?.value; diff --git a/open-sse/translator/concerns/kiroConversation.js b/open-sse/translator/concerns/kiroConversation.js index 11d49dc7..b19e9ba6 100644 --- a/open-sse/translator/concerns/kiroConversation.js +++ b/open-sse/translator/concerns/kiroConversation.js @@ -5,6 +5,20 @@ import { } from "../../config/kiroConstants.js"; const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/; + +/** + * Kiro rejects user turns with empty `content`, so a turn that only carries + * tool results needs placeholder text. It must not read like a user + * instruction: with "continue", models answer the word itself ("Nothing in + * progress to continue") and drop the task they were in the middle of. + */ +export const KIRO_TOOL_RESULTS_PLACEHOLDER = "Tool results provided."; +export const KIRO_EMPTY_USER_PLACEHOLDER = "continue"; + +/** Placeholder content for a user turn with no text of its own. */ +export function kiroEmptyUserContent(hasToolResults) { + return hasToolResults ? KIRO_TOOL_RESULTS_PLACEHOLDER : KIRO_EMPTY_USER_PLACEHOLDER; +} const TOOL_NAME_PATTERN = /[^a-zA-Z0-9_-]/g; function clone(value) { @@ -34,7 +48,6 @@ function uniqueName(rawName, index, usedNames) { const cleaned = String(rawName || "") .trim() .replace(TOOL_NAME_PATTERN, "_") - .replace(/_+/g, "_") .replace(/^_+|_+$/g, ""); const base = trimCodePoints(cleaned || `tool_${index + 1}`, KIRO_TOOL_NAME_MAX_LENGTH); let candidate = base; @@ -174,7 +187,8 @@ function normalizeTurns(history, currentMessage, modelId) { for (const turn of turns) { if (turn.userInputMessage) { - turn.userInputMessage.content = text(turn.userInputMessage.content).trim() || "continue"; + turn.userInputMessage.content = text(turn.userInputMessage.content).trim() + || kiroEmptyUserContent(turn.userInputMessage.userInputMessageContext?.toolResults?.length > 0); turn.userInputMessage.modelId ||= modelId; if (turn.userInputMessage.userInputMessageContext?.tools) { delete turn.userInputMessage.userInputMessageContext.tools; diff --git a/open-sse/translator/concerns/paramSupport.js b/open-sse/translator/concerns/paramSupport.js index e222b23f..863b627e 100644 --- a/open-sse/translator/concerns/paramSupport.js +++ b/open-sse/translator/concerns/paramSupport.js @@ -14,6 +14,9 @@ const STRIP_RULES = [ { provider: "github", match: (m) => /claude/i.test(m) && !/claude.*(opus|sonnet).*4\.6/i.test(m), drop: ["thinking", "reasoning_effort"] }, // Cloudflare Workers AI: content must be plain string, rejects OpenAI content-part array (#1926) { provider: "cloudflare-ai", flattenContent: true }, + // MiMo Desktop Preview models (account-service route): content must be plain string, + // rejects OpenAI content-part array. Cloud models keep their parts (mimo-v2-omni is multi-modal). + { provider: "xiaomi-mimo", match: /preview/i, flattenContent: true }, { provider: "volcengine-ark", match: /glm-5/i, clampToModelMaxOutput: true }, // VolcEngine Ark caps the Kimi family at max_tokens <= 32768, but the model's // advertised ceiling is far higher (Kimi-K2.7-Code resolves to maxOutput 262144), diff --git a/open-sse/translator/concerns/prefetch.js b/open-sse/translator/concerns/prefetch.js index 5055be3c..4d606ebd 100644 --- a/open-sse/translator/concerns/prefetch.js +++ b/open-sse/translator/concerns/prefetch.js @@ -8,6 +8,7 @@ import { fetchImageAsBase64, parseDataUri } from "./image.js"; const TARGETS_NEED_BASE64 = new Set([ FORMATS.GEMINI, FORMATS.GEMINI_CLI, FORMATS.VERTEX, FORMATS.ANTIGRAVITY, FORMATS.OLLAMA, FORMATS.KIRO, + FORMATS.COMMANDCODE, ]); function isRemoteUrl(url) { diff --git a/open-sse/translator/concerns/thinkingUnified.js b/open-sse/translator/concerns/thinkingUnified.js index a3035823..4bc9601f 100644 --- a/open-sse/translator/concerns/thinkingUnified.js +++ b/open-sse/translator/concerns/thinkingUnified.js @@ -19,6 +19,7 @@ const FORMAT_TO_NATIVE = { vertex: "gemini-budget", antigravity: "gemini-budget", kiro: "kiro", + commandcode: "commandcode", }; // Strip a trailing thinking suffix "model(value)" → "model" (no-op when absent). @@ -108,6 +109,7 @@ export const captureThinking = extractThinking; const NATIVE_ONLY_FORMATS = new Set(["gemini-level", "gemini-budget", "claude-budget", "claude-adaptive", "kiro"]); function resolveFormat(targetFormat, model, provider) { + if (targetFormat === "commandcode") return "commandcode"; const providerFmt = provider ? PROVIDERS[provider]?.thinkingFormat : null; if (providerFmt) return providerFmt; const caps = getCapabilitiesForModel(provider, model); @@ -223,10 +225,14 @@ function stripAll(body) { delete body.output_config; if (body.generationConfig) delete body.generationConfig.thinkingConfig; if (body.request?.generationConfig) delete body.request.generationConfig.thinkingConfig; + if (body.params && typeof body.params === "object") { + delete body.params.reasoning_effort; + delete body.params.thinking; + } } // Apply unified thinking config to body in the resolved provider-native format. -function applyFormat(fmt, body, cfg, caps, supportedLevels) { +function applyFormat(fmt, body, cfg, caps, supportedLevels, display) { const none = cfg.mode === "none"; const canDisable = caps.thinkingCanDisable !== false; // Model cannot disable thinking → clamp "none" to minimal effort instead. @@ -243,16 +249,16 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) { if (none && canDisable) { body.thinking = { type: "disabled" }; break; } // Models that can disable thinking need the explicit adaptive switch. // Permanently adaptive models such as Fable 5.1 accept effort directly. - if (canDisable) body.thinking = { type: "adaptive" }; + if (canDisable) body.thinking = { type: "adaptive", ...(display ? { display } : {}) }; else delete body.thinking; const level = toLevel(eff); - body.output_config = { effort: level === "xhigh" ? "high" : level }; + body.output_config = { effort: level === "xhigh" || level === "auto" ? "high" : level }; break; } case "claude-budget": { if (none && canDisable) { body.thinking = { type: "disabled" }; break; } const budget = toBudget(eff, caps.thinkingRange); - body.thinking = budget === -1 ? { type: "enabled" } : { type: "enabled", budget_tokens: budget || 8192 }; + body.thinking = budget === -1 ? { type: "enabled", ...(display ? { display } : {}) } : { type: "enabled", budget_tokens: budget || 8192, ...(display ? { display } : {}) }; break; } case "gemini-level": { @@ -336,6 +342,17 @@ function applyFormat(fmt, body, cfg, caps, supportedLevels) { case "kiro": // Kiro thinking handled via system-tag injection in openai-to-kiro.js; no body field here. break; + case "commandcode": { + // Native CLI sends reasoning_effort inside params of the /alpha/generate envelope. + if (!body.params || typeof body.params !== "object") body.params = {}; + if (none && canDisable) { + delete body.params.reasoning_effort; + break; + } + const level = toLevel(eff); + if (level) body.params.reasoning_effort = level; + break; + } default: break; } @@ -361,7 +378,10 @@ export function applyThinking(targetFormat, model, body, provider = null, intent const fmt = resolveFormat(targetFormat, cleanModel, provider); const supportedLevels = getThinkingLevels(provider, cleanModel); + // Anthropic's `display` (summarized | omitted) decides whether thinking text + // comes back at all; keep what the client asked for instead of resetting it. + const display = typeof body.thinking?.display === "string" ? body.thinking.display : undefined; stripAll(body); - applyFormat(fmt, body, cfg, caps, supportedLevels); + applyFormat(fmt, body, cfg, caps, supportedLevels, display); return body; } diff --git a/open-sse/translator/concerns/toolCall.js b/open-sse/translator/concerns/toolCall.js index 958764dd..251850c1 100644 --- a/open-sse/translator/concerns/toolCall.js +++ b/open-sse/translator/concerns/toolCall.js @@ -1,5 +1,7 @@ // Tool call helper functions for translator +import { FORMATS } from "../formats.js"; + // Anthropic tool_use.id must match: ^[a-zA-Z0-9_-]+$ const TOOL_ID_PATTERN = /^[a-zA-Z0-9_-]+$/; @@ -165,3 +167,16 @@ export function defaultClaudeToolType(tools) { return tools.map(tool => tool?.type ? tool : { ...tool, type: "custom" }); } +// Whether Claude-format tools need explicit `type` defaulting before dispatch. +// Only gateways that declare the `requireClaudeToolType` quirk (MiniMax) reject typeless +// tools. Applying the default globally breaks Claude-format endpoints that only accept the +// legacy typeless tool shape — DeepSeek's Anthropic-compatible endpoint answers HTTP 400 +// "unknown variant `custom`" and every Claude Code request routed there fails (#3905). +export function shouldDefaultClaudeToolType(provider, finalFormat, tools, PROVIDERS) { + return ( + finalFormat === FORMATS.CLAUDE + && Array.isArray(tools) + && PROVIDERS?.[provider]?.quirks?.requireClaudeToolType === true + ); +} + diff --git a/open-sse/translator/formats/claude.js b/open-sse/translator/formats/claude.js index 62a0531a..14c9fc10 100644 --- a/open-sse/translator/formats/claude.js +++ b/open-sse/translator/formats/claude.js @@ -27,6 +27,14 @@ export function lastCacheableToolIndex(tools) { // Check if message has valid non-empty content export function hasValidContent(msg) { if (typeof msg.content === "string" && msg.content.trim()) return true; + if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) { + const block = msg.content; + return !!((block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) || + block.type === CLAUDE_BLOCK.TOOL_USE || + block.type === CLAUDE_BLOCK.TOOL_RESULT || + block.type === CLAUDE_BLOCK.IMAGE || + block.type === CLAUDE_BLOCK.DOCUMENT); + } if (Array.isArray(msg.content)) { return msg.content.some(block => (block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) || @@ -38,6 +46,60 @@ export function hasValidContent(msg) { } return false; } +// Content may arrive as a single content block object (spec allows string | array; +// some clients send the bare object). Wrap it as a one-block array and strip any +// client-placed cache_control: a bare-object marker must never survive +// normalization, on any path, guard or no guard. +function normalizeMessageContent(msg) { + const c = msg?.content; + if (c && typeof c === "object" && !Array.isArray(c)) { + delete c.cache_control; + msg.content = [c]; + } + return msg; +} + +// Total blocks carrying cache_control across system, tools, and messages — the +// upstream Messages API allows at most 4 markers per request. +function countCacheControlBlocks(body) { + let n = 0; + if (Array.isArray(body?.system)) for (const b of body.system) if (b?.cache_control) n++; + if (Array.isArray(body?.tools)) for (const t of body.tools) if (t?.cache_control) n++; + if (Array.isArray(body?.messages)) { + for (const m of body.messages) { + if (Array.isArray(m?.content)) { + for (const b of m.content) if (b?.cache_control) n++; + } else if (m?.content && typeof m.content === "object" && m.content.cache_control) n++; + } + } + return n; +} +// Trim every marker past the 4-marker budget. The head anchors (last system +// block, last cacheable tool) are held; the remaining slots go to the tail-most +// of the other markers in document order. A plain "keep the last 4 in document +// order" rule would drop the head anchors first — they lead document order, yet +// they are exactly what re-anchoring exists to pin. +function capCacheControlBlocks(body) { + const isHead = (b) => { + const sys = Array.isArray(body?.system) ? body.system : []; + if (sys.length && sys[sys.length - 1] === b) return true; + const tools = Array.isArray(body?.tools) ? body.tools : []; + const lastTool = lastCacheableToolIndex(tools); + return lastTool >= 0 && tools[lastTool] === b; + }; + const marked = []; + if (Array.isArray(body?.system)) for (const b of body.system) if (b?.cache_control) marked.push(b); + if (Array.isArray(body?.tools)) for (const t of body.tools) if (t?.cache_control) marked.push(t); + if (Array.isArray(body?.messages)) { + for (const m of body.messages) { + if (Array.isArray(m?.content)) for (const b of m.content) if (b?.cache_control) marked.push(b); + } + } + const head = marked.filter(isHead); + const rest = marked.filter(b => !isHead(b)); + const keep = Math.max(0, 4 - head.length); + for (const b of rest.slice(0, Math.max(0, rest.length - keep))) delete b.cache_control; +} // Fix tool_use/tool_result ordering for Claude API // 1. Assistant message with tool_use: remove text AFTER tool_use (Claude doesn't allow) @@ -136,8 +198,9 @@ function hasForeignServerToolUseId(block) { // Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject: // 1. thinking.type "adaptive" → unsupported on Haiku // 2. output_config.effort → unsupported on Haiku -// 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed -// 4. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright +// 3. bare content-block objects (content: {block} instead of [{block}]) → wrapped first +// 4. role "system" messages (mid-conversation-system beta) → only top-level system is allowed +// 5. server_tool_use blocks carrying a foreign (non-srvtoolu_) id → rejected outright export function normalizeClaudePassthrough(body, model = "") { if (!body || typeof body !== "object") return body; @@ -152,7 +215,15 @@ export function normalizeClaudePassthrough(body, model = "") { if (Object.keys(body.output_config).length === 0) delete body.output_config; } - // 2. Fold mid-conversation system messages into the neighbouring turn. + // 3. Wrap bare content-block objects as one-element arrays before folding. + // Some clients send content: {block} instead of content: [{block}]; the + // mid-conversation-system fold below assumes the array shape, so it must + // run first — a bare-object neighbor would otherwise be zeroed to []. + if (Array.isArray(body.messages)) { + for (const msg of body.messages) normalizeMessageContent(msg); + } + + // 4. Fold mid-conversation system messages into the neighbouring turn. // Hoisting them into body.system would insert volatile content (token counters, // reminders) ahead of the whole conversation and invalidate the prefix cache on // every request. Folding in place keeps the cached prefix stable. @@ -186,7 +257,7 @@ export function normalizeClaudePassthrough(body, model = "") { body.messages = messages; } - // 3. Drop thinking blocks whose signature is not Claude's (combo mixes models, + // 5. Drop thinking blocks whose signature is not Claude's (combo mixes models, // so foreign signatures leak into history and Anthropic rejects them). const thinkingEnabled = body.thinking?.type === "enabled"; const droppedServerToolUseIds = new Set(); @@ -233,7 +304,7 @@ export function normalizeClaudePassthrough(body, model = "") { } } - // 5. Drop empty text blocks and any message left with no content at all. + // 6. Drop empty text blocks and any message left with no content at all. // Anthropic rejects `messages.N.content` blocks with empty text (400 // "text content blocks must be non-empty"); a message whose blocks were all // stripped above must be dropped, not padded with an empty placeholder. @@ -271,7 +342,22 @@ function markLastCacheableBlock(msg) { // (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail. export function anchorClaudeCache(body) { if (!body || typeof body !== "object") return body; + if (Array.isArray(body.messages)) { + for (const msg of body.messages) normalizeMessageContent(msg); + } + // Invalid markers first, whatever the budget: Anthropic rejects a tool that + // carries BOTH defer_loading and cache_control (#3567). The re-anchor path + // below strips them anyway; the over-budget early return used to forward + // them untouched. + if (Array.isArray(body.tools)) { + for (const t of body.tools) { + if (t?.defer_loading === true) delete t.cache_control; + } + } + // Head anchors first, before any budget guard: the 1h TTL on system/tools is + // the point of re-anchoring, and skipping it because the client spent its + // budget would silently downgrade a cache hit to the 5m default. if (Array.isArray(body.system)) { const last = body.system.length - 1; body.system.forEach((block, i) => { @@ -289,6 +375,15 @@ export function anchorClaudeCache(body) { }); } + // Budget guard AFTER the head anchors: with the last system block and last + // tool pinned, at most 2 slots remain. At >= 4 markers the client has spent + // the rest of the budget and every remaining marker is itself a valid + // breakpoint — re-anchoring the tail could only exceed 4, so trim instead. + if (countCacheControlBlocks(body) >= 4) { + capCacheControlBlocks(body); + return body; + } + if (Array.isArray(body.messages)) { let anchored = null; for (let i = body.messages.length - 1; i >= 0; i--) { @@ -320,6 +415,28 @@ export function anchorClaudeCache(body) { // - Add thinking block for Anthropic endpoint (provider === "claude") // - Fix tool_use/tool_result ordering // - Apply cloaking (billing header + fake user ID) for OAuth tokens +export function hoistToolResultImages(body) { + if (!Array.isArray(body?.messages)) return body; + let touched = false; + const messages = body.messages.map((msg) => { + if (msg?.role !== ROLE.USER || !Array.isArray(msg.content)) return msg; + const hoisted = []; + const content = msg.content.map((block) => { + if (block?.type !== CLAUDE_BLOCK.TOOL_RESULT || !Array.isArray(block.content)) return block; + const images = block.content.filter((c) => c?.type === CLAUDE_BLOCK.IMAGE); + if (!images.length) return block; + const rest = block.content.filter((c) => c?.type !== CLAUDE_BLOCK.IMAGE); + hoisted.push({ type: CLAUDE_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` }, ...images); + return { ...block, content: rest.length ? rest : [{ type: CLAUDE_BLOCK.TEXT, text: "(image attached below)" }] }; + }); + if (!hoisted.length) return msg; + touched = true; + // tool_result blocks must lead a user message; the hoisted image follows them. + return { ...msg, content: [...content, ...hoisted] }; + }); + return touched ? { ...body, messages } : body; +} + export function prepareClaudeRequest(body, provider = null, apiKey = null, connectionId = null, rawHeaders = null, sessionId = null) { // quirk: MiniMax's Claude-compatible endpoint rejects Anthropic's output_config (400 invalid params) if (PROVIDERS[provider]?.quirks?.dropOutputConfig) { @@ -368,6 +485,7 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne // Pass 1: remove cache_control + filter empty messages for (let i = 0; i < len; i++) { const msg = body.messages[i]; + normalizeMessageContent(msg); // Remove cache_control from content blocks if (Array.isArray(msg.content)) { @@ -461,8 +579,21 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne // Strip built-in tools (e.g. web_search_20250305) and normalize to Anthropic-native shape // (drop `type` field, fold `function.{name,description,parameters}`) for non-Anthropic providers if (provider !== "claude") { + // Provider-specific whitelist of Anthropic tool `type` values that the + // upstream actually accepts. When the provider declares it + // (e.g. DeepSeek — only web_search_*), keep only listed types; otherwise + // keep the prior behaviour of dropping every non-function tool, which is + // correct for OpenAI-compatible targets reached through this Claude-format + // pass (their tools get normalized below to function-style). + const supportedTypes = PROVIDERS[provider]?.quirks?.claudeSupportedToolTypes; + const hasWhitelist = Array.isArray(supportedTypes); body.tools = body.tools - .filter(tool => !tool.type || tool.type === "function") + .filter(tool => { + const t = tool?.type; + if (!t || t === "function") return true; + if (hasWhitelist) return supportedTypes.includes(t); + return false; + }) .map(tool => { if (tool.function) { return { @@ -471,6 +602,13 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne input_schema: tool.function.parameters, }; } + // When the provider declared a supportedToolTypes whitelist, keep + // the surviving tools' `type` field intact — the upstream + // Anthropic-compatible endpoint (e.g. DeepSeek) requires it to + // route built-ins like web_search_* correctly. Without a + // whitelist, preserve prior behaviour and strip `type` so the + // tool is normalized to plain Anthropic shape. + if (hasWhitelist) return tool; const { type, ...rest } = tool; return rest; }); @@ -492,6 +630,14 @@ export function prepareClaudeRequest(body, provider = null, apiKey = null, conne } } + // Anthropic itself reads images inside tool_result; other Anthropic-compatible + // endpoints (OpenCode Go, Kimi, DeepSeek, GLM, MiniMax) accept image blocks + // only as user content and silently drop them inside a tool result. Move a + // tool's screenshot out of the result and into the same user turn. + if (provider !== "claude" && !provider?.startsWith("anthropic-compatible")) { + body = hoistToolResultImages(body); + } + // Apply cloaking for OAuth tokens (billing header + fake user ID) // session_id in user_id must match X-Claude-Code-Session-Id for fingerprint consistency if ((provider === "claude" || provider?.startsWith("anthropic-compatible")) && apiKey) { diff --git a/open-sse/translator/formats/gemini.js b/open-sse/translator/formats/gemini.js index 8a965fee..729e4c10 100644 --- a/open-sse/translator/formats/gemini.js +++ b/open-sse/translator/formats/gemini.js @@ -432,3 +432,21 @@ export function cleanJSONSchemaForAntigravity(schema) { return cleaned; } +// Merge adjacent same-role messages, strip empty parts, ensure initial user turn +export function normalizeGeminiContents(contents) { + const out = []; + for (const c of contents || []) { + if (!c?.role || !Array.isArray(c.parts)) continue; + const parts = c.parts.filter(p => p && Object.keys(p).length > 0); + if (parts.length === 0) continue; + const last = out.at(-1); + if (last?.role === c.role) last.parts.push(...parts); + else out.push({ ...c, parts: [...parts] }); + } + if (out.length > 0 && out[0].role !== "user") { + out.unshift({ role: "user", parts: [{ text: "..." }] }); + } + return out; +} + + diff --git a/open-sse/translator/formats/responsesApi.js b/open-sse/translator/formats/responsesApi.js index c41ee470..5454f9f3 100644 --- a/open-sse/translator/formats/responsesApi.js +++ b/open-sse/translator/formats/responsesApi.js @@ -23,6 +23,59 @@ export function normalizeResponsesInput(input) { return null; } +// Strict Responses upstreams reject overlong call_ids with InputValidationError (#393). +export const MAX_RESPONSES_CALL_ID_LEN = 64; + +// Fallback ids share one Date.now() when a batch of items is sanitized in a tight +// loop — a per-process sequence keeps same-millisecond ids unique so +// function_call ↔ function_call_output correlation never collides. +let responsesCallIdSeq = 0; + +export function clampResponsesCallId(id) { + if (typeof id !== "string" || !id) return `call_${Date.now()}_${(responsesCallIdSeq += 1)}`; + return id.length > MAX_RESPONSES_CALL_ID_LEN ? id.substring(0, MAX_RESPONSES_CALL_ID_LEN) : id; +} + +// Single-stringify: objects → JSON once; valid JSON strings pass through untouched; +// anything else (partial fragments, empty) falls back to "{}" instead of +// double-encoding and tripping upstream InputValidationError. +export function coerceResponsesArguments(value) { + if (value === undefined || value === null || value === "") return "{}"; + if (typeof value !== "string") { + try { + return JSON.stringify(value); + } catch { + return "{}"; + } + } + try { + JSON.parse(value); + return value; + } catch { + return "{}"; + } +} + +// function_call_output.output must be a string — never null/object. +export function coerceResponsesOutput(value) { + if (typeof value === "string") return value; + if (value === undefined || value === null) return ""; + if (Array.isArray(value)) { + return value.map((c) => { + try { + return c?.text ?? JSON.stringify(c); + } catch { + return String(c); + } + }).join(""); + } + try { + return JSON.stringify(value); + } catch { + return String(value); + } +} + /** * Convert OpenAI Responses API format to standard chat completions format * Responses API uses: { input: [...], instructions: "..." } diff --git a/open-sse/translator/request/claude-to-kiro.js b/open-sse/translator/request/claude-to-kiro.js index 3cfad109..cffe59ec 100644 --- a/open-sse/translator/request/claude-to-kiro.js +++ b/open-sse/translator/request/claude-to-kiro.js @@ -34,6 +34,7 @@ import { ROLE, CLAUDE_BLOCK } from "../schema/index.js"; import { canonicalizeKiroConversation, normalizeKiroToolSpecs, + kiroEmptyUserContent, } from "../concerns/kiroConversation.js"; /** @@ -53,7 +54,8 @@ function convertClaudeMessagesToKiro(messages, model) { const flushPending = () => { if (currentRole === ROLE.USER) { - const content = pendingUserContent.join("\n\n").trim() || "continue"; + const content = pendingUserContent.join("\n\n").trim() + || kiroEmptyUserContent(pendingToolResults.length > 0); const userMsg = { userInputMessage: { content, modelId: model } }; if (pendingImages.length > 0) { @@ -97,11 +99,21 @@ function convertClaudeMessagesToKiro(messages, model) { if (typeof block.content === "string") { resultContent = block.content; } else if (Array.isArray(block.content)) { + // Images a tool returned (screenshots) ride along as user images; + // Kiro tool results are text-only. + let hasImage = false; + for (const c of block.content) { + if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") { + hasImage = true; + const imageType = c.source.media_type || DEFAULT_IMAGE_MIME; + pendingImages.push({ format: imageType.split("/")[1] || imageType, source: { bytes: c.source.data } }); + } + } resultContent = block.content .filter((c) => c.type === CLAUDE_BLOCK.TEXT) .map((c) => c.text) - .join("\n") || JSON.stringify(block.content); + .join("\n") || (hasImage ? "(image attached)" : JSON.stringify(block.content)); } else if (block.content) { resultContent = JSON.stringify(block.content); } @@ -242,9 +254,9 @@ export function claudeToKiroRequest(model, body, stream, credentials) { ? (credentials?.providerSpecificData?.profileArn || "") : (credentials?.providerSpecificData?.profileArn || resolveDefaultProfileArn(authMethod)); - // Kiro CLI/KAS sends system prompt as top-level `systemPrompt`. Keep a - // content fallback too because the CodeWhisperer surface does not always - // enforce top-level systemPrompt for direct calls. + // The system prompt travels inside the first user turn's content (contentPrefix): + // the CodeWhisperer surface rejects a top-level `systemPrompt` with + // 400 REQUEST_BODY_INVALID, so the value below is only a replay cache key. const timestamp = new Date().toISOString(); const systemPromptParts = []; if (thinkingBudget !== null && !usesNativeGptEffort) { @@ -316,14 +328,11 @@ export function claudeToKiroRequest(model, body, stream, credentials) { conversationState: { chatTriggerType: "MANUAL", conversationId, - agentContinuationId: continuationId, - agentTaskType: "vibe", currentMessage: { userInputMessage, }, history: canonical.history, }, - agentMode: "vibe", }; if (profileArn) payload.profileArn = profileArn; @@ -344,6 +353,13 @@ export function claudeToKiroRequest(model, body, stream, credentials) { enumerable: false, }); + // Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the + // reverse map so tool calls stream back under the client's own names. + const restoredToolNames = new Map(); + for (const [original, sanitized] of nameMap) { + if (original !== sanitized) restoredToolNames.set(sanitized, original); + } + if (restoredToolNames.size) payload._toolNameMap = restoredToolNames; return payload; } diff --git a/open-sse/translator/request/claude-to-openai.js b/open-sse/translator/request/claude-to-openai.js index 3956f828..e75d89b8 100644 --- a/open-sse/translator/request/claude-to-openai.js +++ b/open-sse/translator/request/claude-to-openai.js @@ -142,6 +142,13 @@ function systemReminderText(content) { // Convert single Claude message - returns single message or array of messages function convertClaudeMessage(msg) { + // Some clients send content as a single block object; normalize to the + // one-element array every branch below (the system-reminder fold included) + // expects. Must run BEFORE the role branch: systemReminderText only reads + // arrays and strings, so a bare-object system turn was dropped outright. + if (msg.content && typeof msg.content === "object" && !Array.isArray(msg.content)) { + msg.content = [msg.content]; + } // Mid-conversation system message -> user (per Anthropic placement rules) if (msg.role === ROLE.SYSTEM) { const text = systemReminderText(msg.content); @@ -189,25 +196,41 @@ function convertClaudeMessage(msg) { }); break; - case CLAUDE_BLOCK.TOOL_RESULT: + case CLAUDE_BLOCK.TOOL_RESULT: { let resultContent = ""; + const resultImages = []; if (typeof block.content === "string") { resultContent = block.content; } else if (Array.isArray(block.content)) { - resultContent = block.content - .filter(c => c.type === CLAUDE_BLOCK.TEXT) - .map(c => c.text) - .join("\n") || JSON.stringify(block.content); + for (const c of block.content) { + if (c?.type === CLAUDE_BLOCK.IMAGE && c.source?.type === "base64") { + resultImages.push({ + type: OPENAI_BLOCK.IMAGE_URL, + image_url: { url: encodeDataUri(c.source.media_type, c.source.data) } + }); + } + } + const textOnly = block.content.filter(c => c?.type === CLAUDE_BLOCK.TEXT); + resultContent = textOnly.map(c => c.text).join("\n") + || (resultImages.length ? "" : JSON.stringify(block.content)); } else if (block.content) { resultContent = JSON.stringify(block.content); } - + toolResults.push({ role: ROLE.TOOL, tool_call_id: block.tool_use_id, content: resultContent }); + // The OpenAI tool role is text-only, so a screenshot or any other image a + // tool returned would otherwise vanish. Hand it to the model in the user + // turn that follows the tool messages, tagged with the call it came from. + if (resultImages.length) { + parts.push({ type: OPENAI_BLOCK.TEXT, text: `[Image from tool result ${block.tool_use_id}]` }); + parts.push(...resultImages); + } break; + } } } diff --git a/open-sse/translator/request/openai-responses.js b/open-sse/translator/request/openai-responses.js index cf5bc529..5c719fd2 100644 --- a/open-sse/translator/request/openai-responses.js +++ b/open-sse/translator/request/openai-responses.js @@ -6,12 +6,15 @@ */ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; -import { normalizeResponsesInput } from "../formats/responsesApi.js"; +import { + normalizeResponsesInput, + clampResponsesCallId, + coerceResponsesArguments, + coerceResponsesOutput, +} from "../formats/responsesApi.js"; import { ROLE, OPENAI_BLOCK, RESPONSES_ITEM } from "../schema/index.js"; -// Responses API enforces max 64 chars on call_id (#393) -const MAX_CALL_ID_LEN = 64; -const clampCallId = (id) => (typeof id === "string" && id.length > MAX_CALL_ID_LEN ? id.substring(0, MAX_CALL_ID_LEN) : id); +const MAX_TOOL_NAME_LEN = 128; /** * Convert OpenAI Responses API request to OpenAI Chat Completions format @@ -249,6 +252,23 @@ export function openaiResponsesToOpenAIRequest(model, body, stream, credentials) return result; } +/** + * Extract plain text from a system/developer message for Responses instructions. + * Array content (text parts) is joined; anything else falls back to "" rather + * than leaking "[object Object]" upstream. + */ +function extractInstructionsText(content) { + if (typeof content === "string") return content; + if (Array.isArray(content)) { + return content.map((c) => { + if (typeof c?.text === "string") return c.text; + if (typeof c?.content === "string") return c.content; + return ""; + }).filter(Boolean).join("\n"); + } + return ""; +} + /** * Ensure object schema always has properties field (required by Codex Responses API) */ @@ -327,7 +347,7 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) // Use the first instruction-bearing message as instructions. // OpenAI recommends role="developer" for GPT-5/Codex as the system-level prompt. if (!hasSystemMessage) { - result.instructions = typeof msg.content === "string" ? msg.content : ""; + result.instructions = extractInstructionsText(msg.content); hasSystemMessage = true; } continue; // Skip instruction messages in input @@ -378,26 +398,24 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) // Convert tool calls if (msg.role === ROLE.ASSISTANT && msg.tool_calls) { for (const tc of msg.tool_calls) { + // Skip nameless calls — strict Responses upstreams reject them (#444) + const name = typeof tc.function?.name === "string" ? tc.function.name.trim() : ""; + if (!name) continue; result.input.push({ type: RESPONSES_ITEM.FUNCTION_CALL, - call_id: clampCallId(tc.id), - name: tc.function?.name || "_unknown", - arguments: tc.function?.arguments || "{}" + call_id: clampResponsesCallId(tc.id), + name: name.slice(0, MAX_TOOL_NAME_LEN), + arguments: coerceResponsesArguments(tc.function?.arguments) }); } } // Convert tool results - output must be a string for Responses API if (msg.role === ROLE.TOOL) { - const output = typeof msg.content === "string" - ? msg.content - : Array.isArray(msg.content) - ? msg.content.map(c => c.text || JSON.stringify(c)).join("") - : JSON.stringify(msg.content); result.input.push({ type: RESPONSES_ITEM.FUNCTION_CALL_OUTPUT, - call_id: clampCallId(msg.tool_call_id), - output + call_id: clampResponsesCallId(msg.tool_call_id), + output: coerceResponsesOutput(msg.content) }); } } @@ -411,16 +429,19 @@ export function openaiToOpenAIResponsesRequest(model, body, stream, credentials) if (body.tools && Array.isArray(body.tools)) { result.tools = body.tools.map(tool => { if (tool.type === OPENAI_BLOCK.FUNCTION) { + // Strict upstreams reject nameless/overlong tool declarations + const name = typeof tool.function?.name === "string" ? tool.function.name.trim() : ""; + if (!name) return null; return { type: OPENAI_BLOCK.FUNCTION, - name: tool.function.name, + name: name.slice(0, MAX_TOOL_NAME_LEN), description: String(tool.function.description || ""), parameters: normalizeToolParameters(tool.function.parameters), strict: tool.function.strict }; } return tool; - }); + }).filter(Boolean); } // Pass through other relevant fields diff --git a/open-sse/translator/request/openai-to-commandcode.js b/open-sse/translator/request/openai-to-commandcode.js index 9825048b..ac9067f6 100644 --- a/open-sse/translator/request/openai-to-commandcode.js +++ b/open-sse/translator/request/openai-to-commandcode.js @@ -5,6 +5,7 @@ * - params.system: STRING at top level (Anthropic-style; system messages NOT allowed in messages[]) * - params.messages[*].role ∈ {"user","assistant","tool"} * - params.messages[*].content: Array of content blocks (NEVER a string) + * - image_url / image source → {type:"image", image:"data:...;base64,...", mimeType} * - tool_use blocks (assistant): {type:"tool-call", toolCallId, toolName, input} * - tool_result blocks (role=user): {type:"tool-result", toolCallId, toolName, output} * - tools[*]: Anthropic plain {name, description, input_schema} @@ -12,8 +13,9 @@ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; import { randomUUID } from "crypto"; -import { ROLE, OPENAI_BLOCK } from "../schema/index.js"; +import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js"; import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js"; +import { parseDataUri, encodeDataUri } from "../concerns/image.js"; function flattenText(content) { if (content == null) return ""; @@ -29,6 +31,47 @@ function flattenText(content) { return String(content); } +function toNativeImageBlock(part) { + if (!part || typeof part !== "object") return null; + + if (part.type === OPENAI_BLOCK.IMAGE_URL) { + const url = typeof part.image_url === "string" ? part.image_url : part.image_url?.url; + const parsed = parseDataUri(url); + if (!parsed) return null; + return { + type: OPENAI_BLOCK.IMAGE, + image: encodeDataUri(parsed.mimeType, parsed.base64), + mimeType: parsed.mimeType, + mediaType: parsed.mimeType, + }; + } + + if (part.type === OPENAI_BLOCK.IMAGE || part.type === CLAUDE_BLOCK.IMAGE) { + if (typeof part.image === "string" && part.image.startsWith("data:")) { + const parsed = parseDataUri(part.image); + const mime = part.mimeType || parsed?.mimeType || "image/png"; + return { + type: OPENAI_BLOCK.IMAGE, + image: part.image, + mimeType: mime, + mediaType: mime, + }; + } + const source = part.source; + if (source?.type === "base64" && typeof source.data === "string") { + const mime = source.media_type || "image/png"; + return { + type: OPENAI_BLOCK.IMAGE, + image: encodeDataUri(mime, source.data), + mimeType: mime, + mediaType: mime, + }; + } + } + + return null; +} + function toContentBlocks(content) { if (content == null) return [{ type: OPENAI_BLOCK.TEXT, text: "" }]; if (typeof content === "string") return [{ type: OPENAI_BLOCK.TEXT, text: content }]; @@ -40,10 +83,12 @@ function toContentBlocks(content) { } else if (part && typeof part === "object") { if (part.type === OPENAI_BLOCK.TEXT && typeof part.text === "string") { blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); - } else if (part.type === OPENAI_BLOCK.IMAGE_URL || part.type === OPENAI_BLOCK.IMAGE) { - blocks.push({ type: OPENAI_BLOCK.TEXT, text: "[image omitted]" }); - } else if (typeof part.text === "string") { - blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); + } else { + const image = toNativeImageBlock(part); + if (image) blocks.push(image); + else if (typeof part.text === "string") { + blocks.push({ type: OPENAI_BLOCK.TEXT, text: part.text }); + } } } } @@ -88,6 +133,10 @@ function convertMessages(messages = []) { if (role === ROLE.ASSISTANT) { const blocks = []; + const rc = m.reasoning_content || m.thought || m.reasoning; + if (rc || (Array.isArray(m.tool_calls) && m.tool_calls.length > 0)) { + blocks.push({ type: "reasoning", text: rc || " " }); + } const text = flattenText(m.content); if (text) blocks.push({ type: OPENAI_BLOCK.TEXT, text }); if (Array.isArray(m.tool_calls)) { diff --git a/open-sse/translator/request/openai-to-gemini.js b/open-sse/translator/request/openai-to-gemini.js index 9e029e2c..305eaca0 100644 --- a/open-sse/translator/request/openai-to-gemini.js +++ b/open-sse/translator/request/openai-to-gemini.js @@ -2,6 +2,7 @@ import { register } from "../index.js"; import { FORMATS } from "../formats.js"; import { DEFAULT_THINKING_AG_SIGNATURE, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } from "../../config/defaultThinkingSignature.js"; import { openaiToClaudeRequestForAntigravity } from "./openai-to-claude.js"; +import { getGeminiThoughtSignatureSync } from "../../services/thoughtSignatureStore.js"; function generateUUID() { return crypto.randomUUID(); } @@ -14,7 +15,8 @@ import { generateRequestId, generateSessionId, generateProjectId, - cleanJSONSchemaForAntigravity + cleanJSONSchemaForAntigravity, + normalizeGeminiContents } from "../formats/gemini.js"; import { deriveSessionId, toNumericSessionId } from "../../utils/sessionManager.js"; import { ROLE, GEMINI_ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js"; @@ -34,19 +36,8 @@ function sanitizeGeminiFunctionName(name) { return sanitized.substring(0, 64); } -function normalizeGeminiContents(contents) { - const out = []; - for (const c of contents || []) { - if (!c?.role || !Array.isArray(c.parts) || c.parts.length === 0) continue; - const last = out.at(-1); - if (last?.role === c.role) last.parts.push(...c.parts); - else out.push({ ...c, parts: [...c.parts] }); - } - return out; -} - // Core: Convert OpenAI request to Gemini format (base for all variants) -function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE) { +function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG_SIGNATURE, sessionId = null) { const result = { model: model, contents: [], @@ -133,18 +124,27 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG if (msg.tool_calls && Array.isArray(msg.tool_calls)) { const toolCallIds = []; + let firstFunctionCallSeen = false; for (const tc of msg.tool_calls) { if (tc.type !== OPENAI_BLOCK.FUNCTION) continue; const args = tryParseJSON(tc.function?.arguments || "{}"); - parts.push({ - thoughtSignature: signature, + const cachedSig = tc.id ? getGeminiThoughtSignatureSync(tc.id, sessionId, model) : null; + // First call gets cached signature or fallback; sibling calls remain unsigned if no cached sig + const callSig = cachedSig || (!firstFunctionCallSeen ? signature : undefined); + firstFunctionCallSeen = true; + + const part = { functionCall: { id: tc.id, name: sanitizeGeminiFunctionName(tc.function.name), args: args } - }); + }; + if (callSig) { + part.thoughtSignature = callSig; + } + parts.push(part); toolCallIds.push(tc.id); } @@ -153,12 +153,14 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG } // Check if there are actual tool responses in the next messages - const hasActualResponses = toolCallIds.some(fid => toolResponses[fid]); + const isIntermediate = i < body.messages.length - 1; + const hasActualResponses = toolCallIds.some(fid => toolResponses[fid] !== undefined); - if (hasActualResponses) { + if (hasActualResponses || isIntermediate) { const toolParts = []; for (const fid of toolCallIds) { - if (!toolResponses[fid]) continue; + let resp = toolResponses[fid]; + if (resp === undefined) resp = ""; let name = tcID2Name[fid]; if (!name) { @@ -170,7 +172,6 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG } } - let resp = toolResponses[fid]; let parsedResp = tryParseJSON(resp); if (parsedResp === null) { parsedResp = { result: resp }; @@ -232,13 +233,13 @@ function openaiToGeminiBase(model, body, stream, signature = DEFAULT_THINKING_AG } // OpenAI -> Gemini (standard API) -export function openaiToGeminiRequest(model, body, stream) { - return openaiToGeminiBase(model, body, stream); +export function openaiToGeminiRequest(model, body, stream, credentials = null) { + return openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_AG_SIGNATURE, credentials?._clientSessionId); } // OpenAI -> Gemini CLI (Cloud Code Assist) -export function openaiToGeminiCLIRequest(model, body, stream) { - const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE); +export function openaiToGeminiCLIRequest(model, body, stream, credentials = null) { + const gemini = openaiToGeminiBase(model, body, stream, DEFAULT_THINKING_GEMINI_CLI_SIGNATURE, credentials?._clientSessionId); // Thinking is normalized centrally by applyThinking (thinkingUnified.js) after translation. // Clean schema for tools @@ -335,18 +336,26 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu const parts = []; if (Array.isArray(msg.content)) { + let firstToolUseSeen = false; for (const block of msg.content) { if (block.type === CLAUDE_BLOCK.TEXT) { parts.push({ text: block.text }); } else if (block.type === CLAUDE_BLOCK.TOOL_USE) { - parts.push({ - thoughtSignature: signature, + const cachedSig = block.id ? getGeminiThoughtSignatureSync(block.id, credentials?._clientSessionId, model) : null; + const callSig = cachedSig || (!firstToolUseSeen ? signature : undefined); + firstToolUseSeen = true; + + const part = { functionCall: { id: block.id, name: sanitizeGeminiFunctionName(block.name), args: block.input || {} } - }); + }; + if (callSig) { + part.thoughtSignature = callSig; + } + parts.push(part); } else if (block.type === CLAUDE_BLOCK.TOOL_RESULT) { let content = block.content; if (Array.isArray(content)) { diff --git a/open-sse/translator/request/openai-to-kiro.js b/open-sse/translator/request/openai-to-kiro.js index 1d9bedec..b8846660 100644 --- a/open-sse/translator/request/openai-to-kiro.js +++ b/open-sse/translator/request/openai-to-kiro.js @@ -23,6 +23,7 @@ import { ROLE, OPENAI_BLOCK, CLAUDE_BLOCK } from "../schema/index.js"; import { canonicalizeKiroConversation, normalizeKiroToolSpecs, + kiroEmptyUserContent, } from "../concerns/kiroConversation.js"; /** @@ -51,7 +52,8 @@ function convertMessages(messages, model) { const flushPending = () => { if (currentRole === "user") { - const content = pendingUserContent.join("\n\n").trim() || "continue"; + const content = pendingUserContent.join("\n\n").trim() + || kiroEmptyUserContent(pendingToolResults.length > 0); const userMsg = { userInputMessage: { content: content, @@ -340,9 +342,9 @@ export function openaiToKiroRequest(model, body, stream, credentials) { const timestamp = new Date().toISOString(); - // Kiro CLI/KAS sends these as top-level systemPrompt. Keep a content fallback - // too because the CodeWhisperer surface does not always enforce top-level - // systemPrompt for direct calls. + // The system prompt travels inside the first user turn's content (contentPrefix): + // the CodeWhisperer surface rejects a top-level `systemPrompt` with + // 400 REQUEST_BODY_INVALID, so the value below is only a replay cache key. const systemPromptParts = []; if (thinkingBudget !== null && !usesNativeGptEffort) { systemPromptParts.push(buildThinkingSystemPrefix(thinkingBudget)); @@ -397,8 +399,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) { conversationState: { chatTriggerType: "MANUAL", conversationId, - agentContinuationId: continuationId, - agentTaskType: "vibe", currentMessage: { userInputMessage: { content: replayCurrent.content || "", @@ -414,7 +414,6 @@ export function openaiToKiroRequest(model, body, stream, credentials) { }, history: canonical.history }, - agentMode: "vibe", }; if (profileArn) { @@ -437,6 +436,13 @@ export function openaiToKiroRequest(model, body, stream, credentials) { enumerable: false }); + // Kiro tool specs get sanitized names (`mcp__a__b` → `mcp_a_b`); keep the + // reverse map so tool calls stream back under the client's own names. + const restoredToolNames = new Map(); + for (const [original, sanitized] of nameMap) { + if (original !== sanitized) restoredToolNames.set(sanitized, original); + } + if (restoredToolNames.size) payload._toolNameMap = restoredToolNames; return payload; } diff --git a/open-sse/translator/response/commandcode-to-openai.js b/open-sse/translator/response/commandcode-to-openai.js index ab3d7d7b..b75a2e70 100644 --- a/open-sse/translator/response/commandcode-to-openai.js +++ b/open-sse/translator/response/commandcode-to-openai.js @@ -165,12 +165,11 @@ export function commandCodeToOpenAIResponse(chunk, state) { break; } case "error": { - state.finishReason = OPENAI_FINISH.STOP; const errVal = event.error ?? event.message ?? "unknown"; const errStr = typeof errVal === "string" ? errVal : JSON.stringify(errVal); - out.push(makeChunk(state, { content: `\n\n[CommandCode error: ${errStr}]` })); - out.push(makeChunk(state, {}, OPENAI_FINISH.STOP)); - break; + // Mid-stream error: throw rather than emitting as fake content with finish_reason: "stop" + // This ensures the downstream stream handler marks the stream as errored/aborted. + throw new Error(`[CommandCode error: ${errStr}]`); } // Silently ignore: start, start-step, reasoning-start, reasoning-end, text-start, text-end, // provider-metadata, message-metadata, etc. They carry no client-visible content. diff --git a/open-sse/translator/response/gemini-to-openai.js b/open-sse/translator/response/gemini-to-openai.js index 8f04ffd3..14c47939 100644 --- a/open-sse/translator/response/gemini-to-openai.js +++ b/open-sse/translator/response/gemini-to-openai.js @@ -6,6 +6,7 @@ import { toOpenAIUsage } from "../concerns/usage.js"; import { reasoningDelta } from "../concerns/reasoning.js"; import { encodeDataUri } from "../concerns/image.js"; import { toOpenAIFinish } from "../concerns/finishReason.js"; +import { storeGeminiThoughtSignature } from "../../services/thoughtSignatureStore.js"; // Build chunk meta for current gemini state function chunkMeta(state) { @@ -13,14 +14,18 @@ function chunkMeta(state) { } // Build a tool_call chunk from a gemini functionCall part (shared by sig/non-sig branches) -function emitFunctionCall(functionCall, state) { +function emitFunctionCall(functionCall, state, signature = null) { const rawName = functionCall.name; // Restore original tool name from mapping (AG cloaking) const fcName = state.toolNameMap?.get(rawName) || rawName; const fcArgs = functionCall.args || {}; const toolCallIndex = state.functionIndex++; + const callId = functionCall.id || `${fcName}-${Date.now()}-${toolCallIndex}`; + if (signature) { + storeGeminiThoughtSignature(callId, signature, state.sessionId, state.model); + } const toolCall = { - id: `${fcName}-${Date.now()}-${toolCallIndex}`, + id: callId, index: toolCallIndex, type: OPENAI_BLOCK.FUNCTION, function: { name: fcName, arguments: JSON.stringify(fcArgs) }, @@ -47,7 +52,7 @@ export function geminiToOpenAIResponse(chunk, state) { // Initialize state if (!state.messageId) { state.messageId = response.responseId || `msg_${Date.now()}`; - state.model = response.modelVersion || "gemini"; + state.model = response.modelVersion || state.model || "gemini"; state.functionIndex = 0; state.geminiToolCallCount = 0; results.push(buildChunk(chunkMeta(state), { role: ROLE.ASSISTANT }, null)); @@ -57,13 +62,21 @@ export function geminiToOpenAIResponse(chunk, state) { if (content?.parts) { for (const part of content.parts) { const hasThoughtSig = part.thoughtSignature || part.thought_signature; + if (hasThoughtSig && typeof hasThoughtSig === "string") { + state.pendingThoughtSignature = hasThoughtSig; + } const isThought = part.thought === true; - + // Handle thought signature (thinking mode) if (hasThoughtSig) { const hasTextContent = part.text !== undefined && part.text !== ""; const hasFunctionCall = !!part.functionCall; - + + // Standalone thoughtSignature part (no text, no functionCall): keep pending for next functionCall + if (!hasTextContent && !hasFunctionCall) { + continue; + } + if (hasTextContent) { results.push(buildChunk( chunkMeta(state), @@ -71,9 +84,10 @@ export function geminiToOpenAIResponse(chunk, state) { null )); } - + if (hasFunctionCall) { - results.push(emitFunctionCall(part.functionCall, state)); + results.push(emitFunctionCall(part.functionCall, state, hasThoughtSig)); + state.pendingThoughtSignature = null; } continue; } @@ -92,7 +106,9 @@ export function geminiToOpenAIResponse(chunk, state) { // Function call if (part.functionCall) { - results.push(emitFunctionCall(part.functionCall, state)); + const sig = state.pendingThoughtSignature || null; + results.push(emitFunctionCall(part.functionCall, state, sig)); + state.pendingThoughtSignature = null; } // Inline data (images) diff --git a/open-sse/translator/response/kiro-to-claude.js b/open-sse/translator/response/kiro-to-claude.js index 455672b1..d9fc0aab 100644 --- a/open-sse/translator/response/kiro-to-claude.js +++ b/open-sse/translator/response/kiro-to-claude.js @@ -46,6 +46,14 @@ function convertFinishReason(reason) { * Convert one OpenAI-format chunk (from KiroExecutor) into Claude SSE events. * Returns an array of Claude events, or null when the chunk yields nothing. */ +// Kiro only accepts sanitized tool names; the request translator leaves the +// reverse map on the stream state so calls come back under the client's names. +function restoreToolName(stateOrData, name) { + const raw = name || ""; + const map = stateOrData?.toolNameMap || stateOrData?._toolNameMap; + return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw; +} + export function kiroToClaudeResponse(chunk, state) { // KiroExecutor emits chat.completion.chunk objects; tolerate string chunks // by attempting a parse (defensive — the direct path is always objects). @@ -161,7 +169,7 @@ export function kiroToClaudeResponse(chunk, state) { const toolBlockIndex = state.nextBlockIndex++; state.toolCalls.set(idx, { id: tc.id, - name: tc.function?.name || "", + name: restoreToolName(state, tc.function?.name), blockIndex: toolBlockIndex, }); results.push({ @@ -170,7 +178,7 @@ export function kiroToClaudeResponse(chunk, state) { content_block: { type: "tool_use", id: tc.id, - name: tc.function?.name || "", + name: restoreToolName(state, tc.function?.name), input: {}, }, }); @@ -246,7 +254,7 @@ export function kiroToClaudeNonStreaming(data) { content.push({ type: "tool_use", id: tc.id || `toolu_${Date.now()}`, - name: tc.function?.name || "", + name: restoreToolName(data, tc.function?.name), input, }); } diff --git a/open-sse/translator/response/kiro-to-openai.js b/open-sse/translator/response/kiro-to-openai.js index 7059a851..8e713fd7 100644 --- a/open-sse/translator/response/kiro-to-openai.js +++ b/open-sse/translator/response/kiro-to-openai.js @@ -20,13 +20,38 @@ function chunkMeta(state) { * Parse Kiro SSE event and convert to OpenAI format * Kiro events: assistantResponseEvent, codeEvent, supplementaryWebLinksEvent, etc. */ +// Kiro only accepts sanitized tool names; the request translator leaves the +// reverse map on the stream state so calls come back under the client's names. +function restoreToolName(state, name) { + const raw = name || ""; + const map = state?.toolNameMap; + return map && typeof map.get === "function" && map.has(raw) ? map.get(raw) : raw; +} + export function kiroToOpenAIResponse(chunk, state) { if (!chunk) return null; - // If chunk is already in OpenAI format (from executor transform), return as-is + // If chunk is already in OpenAI format (from executor transform), return it + // with the client's tool names restored. if (chunk.object === "chat.completion.chunk" && chunk.choices) { - return chunk; + if (!state?.toolNameMap?.size) return chunk; + return { + ...chunk, + choices: chunk.choices.map((choice) => { + const calls = choice?.delta?.tool_calls; + if (!Array.isArray(calls)) return choice; + return { + ...choice, + delta: { + ...choice.delta, + tool_calls: calls.map((tc) => tc?.function?.name + ? { ...tc, function: { ...tc.function, name: restoreToolName(state, tc.function.name) } } + : tc), + }, + }; + }), + }; } // Handle string chunk (raw SSE data) @@ -109,7 +134,7 @@ export function kiroToOpenAIResponse(chunk, state) { state.hadToolUse = true; const toolUse = data.toolUseEvent || data; const toolCallId = toolUse.toolUseId || fallbackToolCallId(); - const toolName = toolUse.name || ""; + const toolName = restoreToolName(state, toolUse.name); const toolInput = toolUse.input || {}; const openaiChunk = buildChunk(chunkMeta(state), { diff --git a/open-sse/translator/response/openai-responses.js b/open-sse/translator/response/openai-responses.js index ff55bb4e..bd435f9c 100644 --- a/open-sse/translator/response/openai-responses.js +++ b/open-sse/translator/response/openai-responses.js @@ -446,6 +446,13 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { state.created = Math.floor(Date.now() / 1000); state.toolCallIndex = 0; state.currentToolCallId = null; + // item_id → chat tool_calls index. Deltas carry item_id; keying on it (not + // stream position) keeps parallel calls separate when upstream emits all + // output_item.added events before any done/delta. Lazily created so callers + // that build their own state object (stream.js) need no changes. + state.respToolChatIndex ??= new Map(); + // Indices that already received argument deltas (guards done-with-args). + state.respToolArgsEmitted ??= new Set(); } // Text content delta @@ -464,16 +471,29 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { return null; } - // Function call started (standard function_call or custom_tool_call) + // Function call started (standard function_call or custom_tool_call). + // Index is assigned here (not on done): attributing deltas by stream position + // merges parallel calls into index 0 whenever upstream emits all addeds + // before dones — the client then concatenates N JSON payloads into one + // tool input and fails validation. The server item id is the correlator. if (eventType === "response.output_item.added" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) { const item = data.item; state.currentToolCallId = item.call_id || fallbackToolCallId(); + state.respToolChatIndex ??= new Map(); + const key = item.id || data.item_id || state.currentToolCallId; + let idx; + if (key && state.respToolChatIndex.has(key)) { + idx = state.respToolChatIndex.get(key); // duplicate added (retry) — reuse + } else { + idx = state.toolCallIndex++; + if (key) state.respToolChatIndex.set(key, idx); + } return buildChunk( { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, { tool_calls: [{ - index: state.toolCallIndex, + index: idx, id: state.currentToolCallId, type: OPENAI_BLOCK.FUNCTION, function: { name: item.name || "", arguments: "" } @@ -482,20 +502,39 @@ export function openaiResponsesToOpenAIResponse(chunk, state) { ); } - // Function call arguments delta (standard or custom_tool_call variant) + // Function call arguments delta (standard or custom_tool_call variant). + // Routed by item_id so interleaved parallel fragments stay on their own call. if (eventType === "response.function_call_arguments.delta" || eventType === "response.custom_tool_call_input.delta") { const argsDelta = data.delta || ""; if (!argsDelta) return null; + const known = data.item_id ? state.respToolChatIndex?.get(data.item_id) : undefined; + const idx = known ?? Math.max(0, (state.toolCallIndex || 1) - 1); + state.respToolArgsEmitted ??= new Set(); + state.respToolArgsEmitted.add(idx); return buildChunk( { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, - { tool_calls: [{ index: state.toolCallIndex, function: { arguments: argsDelta } }] } + { tool_calls: [{ index: idx, function: { arguments: argsDelta } }] } ); } - // Function call done (standard or custom_tool_call variant) + // Function call done (standard or custom_tool_call variant). + // Index was assigned at added-time; nothing to advance. Some upstreams send + // complete arguments only here (no deltas) — emit them once in that case. if (eventType === "response.output_item.done" && (data.item?.type === RESPONSES_ITEM.FUNCTION_CALL || data.item?.type === "custom_tool_call")) { - state.toolCallIndex++; + const key = data.item?.id || data.item_id; + const idx = (key && state.respToolChatIndex?.get(key)) ?? Math.max(0, (state.toolCallIndex || 1) - 1); + const fullArgs = data.item?.arguments; + if (typeof fullArgs === "string" && fullArgs) { + state.respToolArgsEmitted ??= new Set(); + if (!state.respToolArgsEmitted.has(idx)) { + state.respToolArgsEmitted.add(idx); + return buildChunk( + { id: state.chatId, created: state.created, model: state.model || MODEL_FALLBACK }, + { tool_calls: [{ index: idx, function: { arguments: fullArgs } }] } + ); + } + } return null; } diff --git a/open-sse/utils/codexToolSchema.js b/open-sse/utils/codexToolSchema.js new file mode 100644 index 00000000..0c017b61 --- /dev/null +++ b/open-sse/utils/codexToolSchema.js @@ -0,0 +1,80 @@ +// Codex-specific tool JSON Schema compatibility. +// +// `https://chatgpt.com/backend-api/codex/responses` validates every function +// tool's `parameters` with a regex engine that does not implement Unicode +// property escapes. A `pattern` such as +// +// "^(?!__.*__$)[^\\p{Cc}\\p{Cf}\\p{Zl}\\p{Zp}\"\\\\./\\[\\]]{1,200}$" +// +// is a perfectly valid ECMAScript `u`-mode regex, but Codex answers +// +// 400 Invalid schema for function 'Artifact': '^\p{Cc}...' is not a 'regex' +// param: tools[0].parameters +// +// The request is deterministically malformed for this provider, so every +// account fails identically and the combo pays a full failover before landing +// somewhere that accepts it (#3922). +// +// Scope guardrail (#3667): this is NOT a global schema sanitizer. Providers +// that do support `\p{...}` keep the constraint untouched — the strip runs only +// on the Codex dispatch path, and only on `pattern` strings that actually +// contain a property escape. Everything else in the schema (including valid +// patterns) passes through byte-identical. + +// `\p{...}` / `\P{...}` with an odd number of preceding backslashes — an even +// count means the backslash itself is escaped, so `\\p{Cc}` is a literal "p". +const UNICODE_PROPERTY_ESCAPE = /(^|[^\\])(\\\\)*\\[pP]\{/; + +export function hasUnicodePropertyEscape(pattern) { + return typeof pattern === "string" && UNICODE_PROPERTY_ESCAPE.test(pattern); +} + +// Copy-on-write walk: returns the original reference when nothing changed, so +// untouched schemas keep object identity and callers can cheaply detect a no-op. +// `properties` is special-cased because its keys are arbitrary property *names* +// (which may themselves be "pattern" or "properties") and must never be read as +// schema keywords; every other key recurses as an ordinary schema node. +function stripNode(node, stats) { + if (Array.isArray(node)) { + let changed = false; + const next = node.map((item) => { + const cleaned = stripNode(item, stats); + if (cleaned !== item) changed = true; + return cleaned; + }); + return changed ? next : node; + } + if (!node || typeof node !== "object") return node; + + let changed = false; + const next = {}; + for (const [key, value] of Object.entries(node)) { + if (key === "pattern" && hasUnicodePropertyEscape(value)) { + stats.removed++; + changed = true; + continue; + } + if (key === "properties" && value && typeof value === "object" && !Array.isArray(value)) { + let propsChanged = false; + const props = {}; + for (const [propName, propSchema] of Object.entries(value)) { + const cleaned = stripNode(propSchema, stats); + if (cleaned !== propSchema) propsChanged = true; + props[propName] = cleaned; + } + if (propsChanged) changed = true; + next[key] = propsChanged ? props : value; + continue; + } + const cleaned = stripNode(value, stats); + if (cleaned !== value) changed = true; + next[key] = cleaned; + } + return changed ? next : node; +} + +// Remove only the `pattern` constraints Codex's validator rejects. +// Returns the same reference when the schema is already compatible. +export function stripCodexUnsupportedPatterns(schema, stats = { removed: 0 }) { + return stripNode(schema, stats); +} diff --git a/open-sse/utils/stream.js b/open-sse/utils/stream.js index 8fa8c11f..50bac1bc 100644 --- a/open-sse/utils/stream.js +++ b/open-sse/utils/stream.js @@ -49,7 +49,8 @@ export function createSSEStream(options = {}) { connectionId = null, body = null, onStreamComplete = null, - apiKey = null + apiKey = null, + credentials = null } = options; let buffer = ""; @@ -59,7 +60,7 @@ export function createSSEStream(options = {}) { const decoder = new TextDecoder("utf-8", { fatal: false }); const state = mode === STREAM_MODE.TRANSLATE - ? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model } + ? { ...initState(sourceFormat), provider, toolNameMap, customToolNames: new Set(customToolNames || []), model, sessionId: credentials?._clientSessionId || null } : null; let totalContentLength = 0; @@ -485,7 +486,7 @@ export function createSSEStream(options = {}) { }); } -export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null, customToolNames = null) { +export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, provider = null, reqLogger = null, toolNameMap = null, model = null, connectionId = null, body = null, onStreamComplete = null, apiKey = null, customToolNames = null, credentials = null) { return createSSEStream({ mode: STREAM_MODE.TRANSLATE, targetFormat, @@ -498,7 +499,8 @@ export function createSSETransformStreamWithLogger(targetFormat, sourceFormat, p connectionId, body, onStreamComplete, - apiKey + apiKey, + credentials }); } diff --git a/open-sse/utils/streamHandler.js b/open-sse/utils/streamHandler.js index 6846c557..bac14a99 100644 --- a/open-sse/utils/streamHandler.js +++ b/open-sse/utils/streamHandler.js @@ -95,6 +95,9 @@ export function createStreamController({ onDisconnect, onError, log, provider, m * activity), not here — output of the transform stream may be silent * for long periods while raw bytes still flow (e.g. Kiro EventStream * binary frames buffering, Claude reasoning streams). + * + * @param {function} [onAbortTerminal] - Receives a human-readable abort + * message and returns terminal SSE bytes to emit downstream. */ export function createDisconnectAwareStream(transformStream, streamController, onAbortTerminal = null) { const reader = transformStream.readable.getReader(); @@ -191,6 +194,7 @@ export function createDisconnectAwareStream(transformStream, streamController, o */ export function pipeWithDisconnect(providerResponse, transformStream, streamController, onAbortTerminal = null, stallTimeoutMs = STREAM_STALL_TIMEOUT_MS) { let stallTimer = null; + let abortMessage = "upstream connection lost"; const clearStall = () => { if (stallTimer) { clearTimeout(stallTimer); stallTimer = null; } }; @@ -198,6 +202,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont clearStall(); stallTimer = setTimeout(() => { stallTimer = null; + abortMessage = "stream stall timeout"; streamController.handleError?.(new Error("stream stall timeout")); streamController.abort?.(); }, stallTimeoutMs); @@ -233,7 +238,7 @@ export function pipeWithDisconnect(providerResponse, transformStream, streamCont return createDisconnectAwareStream( { readable: transformedBody, writable: { getWriter: () => ({ abort: () => Promise.resolve() }) } }, wrappedController, - onAbortTerminal + onAbortTerminal ? () => onAbortTerminal(abortMessage) : null ); } diff --git a/open-sse/utils/streamHelpers.js b/open-sse/utils/streamHelpers.js index a7a19180..ee4e720a 100644 --- a/open-sse/utils/streamHelpers.js +++ b/open-sse/utils/streamHelpers.js @@ -1,4 +1,8 @@ import { FORMATS } from "../translator/formats.js"; +import { buildErrorBody } from "./error.js"; +import { SSE_DONE } from "./sseConstants.js"; + +const sharedEncoder = new TextEncoder(); // Parse SSE data line export function parseSSELine(line, format = null) { @@ -120,3 +124,24 @@ export function formatSSE(data, sourceFormat) { return `data: ${JSON.stringify(data)}\n\n`; } + +// Terminal frames for a stream that aborted after HTTP 200 was already sent, so +// the status code can no longer change. OpenAI-compatible clients (openai-python +// raises APIError on any `data:` payload carrying an `error` key, checked before +// [DONE]) need the error frame first, then [DONE]; Anthropic clients need +// `event: error`. Never fabricate a successful finish_reason instead. +// +// Returns encoded bytes: onAbortTerminal callbacks are enqueued verbatim, same +// as buildAbortedResponsesTerminalBytes. +// +// NOTE: non-SSE client formats (Ollama NDJSON) get an SSE frame here — dead in +// practice because detectFormatByEndpoint never resolves to OLLAMA. +export function buildStreamErrorBytes(statusCode, message, clientFormat) { + const { error } = buildErrorBody(statusCode, message); + + const sse = clientFormat === FORMATS.CLAUDE + ? formatSSE({ type: "error", error }, FORMATS.CLAUDE) + : formatSSE({ error }, clientFormat) + SSE_DONE; + + return sharedEncoder.encode(sse); +} diff --git a/package-lock.json b/package-lock.json index f39a4ae8..4564afba 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "9router-app", - "version": "1.0.11", + "version": "1.0.13", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "9router-app", - "version": "1.0.11", + "version": "1.0.13", "dependencies": { "@dnd-kit/core": "^6.3.1", "@dnd-kit/modifiers": "^9.0.0", @@ -98,6 +98,7 @@ "integrity": "sha512-RgHBCvtjbOK2gXSNBNIkNoEc9qoVEtau3hj8gEqKQuL3HZAibKarWFEI3Lfm6EYKkLalOh8eSrj9b+ch9H/VBA==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@babel/code-frame": "^7.29.7", "@babel/generator": "^7.29.7", @@ -324,6 +325,7 @@ "resolved": "https://registry.npmjs.org/@dnd-kit/core/-/core-6.3.1.tgz", "integrity": "sha512-xkGBRQQab4RLwgXxoqETICr6S5JlogafbhNsidmrkVv2YRs5MLwpjoF2qpiGjQt8S9AoxtIV603s0GIUpY5eYQ==", "license": "MIT", + "peer": true, "dependencies": { "@dnd-kit/accessibility": "^3.1.1", "@dnd-kit/utilities": "^3.2.2", @@ -374,32 +376,10 @@ "react": ">=16.8.0" } }, - "node_modules/@emnapi/core": { - "version": "1.10.0", - "resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.10.0.tgz", - "integrity": "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==", - "dev": true, - "license": "MIT", - "optional": true, - "dependencies": { - "@emnapi/wasi-threads": "1.2.1", - "tslib": "^2.4.0" - } - }, - "node_modules/@emnapi/runtime": { - "version": "1.11.3", - "resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.11.3.tgz", - "integrity": "sha512-Xz4Tpyki7XyrpbUK1jR1AhdAdaXyhhY4lZ3neLodmhpuWfy2PAQN5B46sAiU4liOXGLkHypn/qU+jvfWSCYYLA==", - "license": "MIT", - "optional": true, - "dependencies": { - "tslib": "^2.4.0" - } - }, "node_modules/@emnapi/wasi-threads": { - "version": "1.2.1", - "resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.1.tgz", - "integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==", + "version": "1.2.3", + "resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.3.tgz", + "integrity": "sha512-ELEBe8PsLvvJ6QMr0zLt8ffvOHW/dc1m3CEzNMg7aJUv3bMaoDtw2TXyDAwkYBuroxxuHEwhRTLJSe5sya547g==", "dev": true, "license": "MIT", "optional": true, @@ -1887,27 +1867,6 @@ "node": ">=14.0.0" } }, - "node_modules/@tailwindcss/oxide-wasm32-wasi/node_modules/@emnapi/core": { - "version": "1.11.1", - "dev": true, - "inBundle": true, - "license": "MIT", - "optional": true, - "dependencies": { - "@emnapi/wasi-threads": "1.2.2", - "tslib": "^2.4.0" - } - }, - "node_modules/@tailwindcss/oxide-wasm32-wasi/node_modules/@emnapi/runtime": { - "version": "1.11.1", - "dev": true, - "inBundle": true, - "license": "MIT", - "optional": true, - "dependencies": { - "tslib": "^2.4.0" - } - }, "node_modules/@tailwindcss/oxide-wasm32-wasi/node_modules/@emnapi/wasi-threads": { "version": "1.2.2", "dev": true, @@ -2245,6 +2204,7 @@ "integrity": "sha512-fUBfTuuEulWqX6V8+O3PtScV01tzYYRUDTAirHFKoRAt7nOzoGiPt0M/bB47wWNy0coOOcgEwAMUtBpykMxl6w==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@typescript-eslint/scope-manager": "8.67.0", "@typescript-eslint/types": "8.67.0", @@ -2766,6 +2726,18 @@ "node": ">=14.0.0" } }, + "node_modules/@unrs/resolver-binding-wasm32-wasi/node_modules/@emnapi/core": { + "version": "1.10.0", + "resolved": "https://registry.npmjs.org/@emnapi/core/-/core-1.10.0.tgz", + "integrity": "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==", + "dev": true, + "license": "MIT", + "optional": true, + "dependencies": { + "@emnapi/wasi-threads": "1.2.1", + "tslib": "^2.4.0" + } + }, "node_modules/@unrs/resolver-binding-wasm32-wasi/node_modules/@emnapi/runtime": { "version": "1.10.0", "resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.10.0.tgz", @@ -2777,6 +2749,17 @@ "tslib": "^2.4.0" } }, + "node_modules/@unrs/resolver-binding-wasm32-wasi/node_modules/@emnapi/wasi-threads": { + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/@emnapi/wasi-threads/-/wasi-threads-1.2.1.tgz", + "integrity": "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==", + "dev": true, + "license": "MIT", + "optional": true, + "dependencies": { + "tslib": "^2.4.0" + } + }, "node_modules/@unrs/resolver-binding-win32-arm64-msvc": { "version": "1.12.2", "resolved": "https://registry.npmjs.org/@unrs/resolver-binding-win32-arm64-msvc/-/resolver-binding-win32-arm64-msvc-1.12.2.tgz", @@ -2926,6 +2909,7 @@ "integrity": "sha512-lGq+9yr1/GuAWaVYIHRjvvySG5/4VfKIvC8EWxStPdcDh/Ka7FG3twP6v4d5BkravUilhIAsG4Qj83t02LWUPQ==", "dev": true, "license": "MIT", + "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -3407,6 +3391,7 @@ } ], "license": "MIT", + "peer": true, "dependencies": { "baseline-browser-mapping": "^2.11.12", "caniuse-lite": "^1.0.30001809", @@ -3822,6 +3807,7 @@ "resolved": "https://registry.npmjs.org/d3-selection/-/d3-selection-3.0.0.tgz", "integrity": "sha512-fmTRWbNMmsmWq6xJV8D19U/gw/bwrHfNXxrIN+HfZgnzqTHp9jOmKMhsTUjXOJnZOdZY9Q28y4yebKzqDKlxlQ==", "license": "ISC", + "peer": true, "engines": { "node": ">=12" } @@ -4449,6 +4435,7 @@ "integrity": "sha512-DgZS62aPLXKlnxILS/AYCoRvHaZeXceIzlXPkkGGzJWSow1aEk0lbTlxUSlyjC8jcaKxAdOnTDz+o1JFSBsyjw==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@eslint-community/eslint-utils": "^4.8.0", "@eslint-community/regexpp": "^4.12.1", @@ -4634,6 +4621,7 @@ "integrity": "sha512-whOE1HFo/qJDyX4SnXzP4N6zOWn79WhnCUY/iDR0mPfQZO8wcYE4JClzI2oZrhBnnMUCBCHZhO6VQyoBU95mZA==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@rtsao/scc": "^1.1.0", "array-includes": "^3.1.9", @@ -5571,6 +5559,7 @@ "resolved": "https://registry.npmjs.org/immer/-/immer-11.1.16.tgz", "integrity": "sha512-Xs7H9rBc+kti1J6RueUvbEBkmOz7jqj11XYgf+YMXAYzu8EeE7hwZ9poLXdVfVnGmJu7QAf41T7H2KuF6QoK6Q==", "license": "MIT", + "peer": true, "funding": { "type": "opencollective", "url": "https://opencollective.com/immer" @@ -6864,6 +6853,7 @@ "resolved": "https://registry.npmjs.org/monaco-editor/-/monaco-editor-0.56.0.tgz", "integrity": "sha512-sXboRm3BeBeLm938eaiyLMe0OxzfXIlZvbv4ir/jVgQy1zDhWjgmny0WoN45fuDKhCCQsYMbBJrv/A6jd8aCUg==", "license": "MIT", + "peer": true, "dependencies": { "dompurify": "3.4.8", "marked": "14.0.0" @@ -6949,6 +6939,7 @@ "resolved": "https://registry.npmjs.org/next/-/next-16.3.1.tgz", "integrity": "sha512-hsAp0i7Rh+/dhe7DGIeN2YlpLM1DP4MNxti9EtDMtqcO612X81MvvEj388/oTce9U1EcEIOWDlGq0zRwrBKvuA==", "license": "MIT", + "peer": true, "dependencies": { "@next/env": "16.3.1", "@swc/helpers": "0.5.23", @@ -7705,6 +7696,7 @@ "resolved": "https://registry.npmjs.org/react/-/react-19.2.4.tgz", "integrity": "sha512-9nfp2hYpCwOjAN+8TZFGhtWEwgvWHXqESH8qT89AT/lWklpLON22Lc8pEtnpsZz7VmawabSU0gCjnj8aC0euHQ==", "license": "MIT", + "peer": true, "engines": { "node": ">=0.10.0" } @@ -7714,6 +7706,7 @@ "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.2.4.tgz", "integrity": "sha512-AXJdLo8kgMbimY95O2aKQqsz2iWi9jMgKJhRBAxECE4IFxfcazB2LmzloIoibJI3C12IlY20+KFaLv+71bUJeQ==", "license": "MIT", + "peer": true, "dependencies": { "scheduler": "^0.27.0" }, @@ -7725,13 +7718,15 @@ "version": "16.13.1", "resolved": "https://registry.npmjs.org/react-is/-/react-is-16.13.1.tgz", "integrity": "sha512-24e6ynE2H+OKt4kqsOvNd8kBpV65zoxbA4BVsEOB3ARVWQki/DHzaUoC5KuON/BiccDaCCTZBuOcfZs70kR8bQ==", - "license": "MIT" + "license": "MIT", + "peer": true }, "node_modules/react-redux": { "version": "9.3.0", "resolved": "https://registry.npmjs.org/react-redux/-/react-redux-9.3.0.tgz", "integrity": "sha512-KQopgqFo/p/fgmAs5qz6p5RWaNAzq40WAu7fJIXnQpYxFPbJYtsJPWvGeF2rOBaY/kEuV77AVsX8TsQzKm+A/g==", "license": "MIT", + "peer": true, "dependencies": { "@types/use-sync-external-store": "^0.0.6", "use-sync-external-store": "^1.4.0" @@ -7805,7 +7800,8 @@ "version": "5.0.1", "resolved": "https://registry.npmjs.org/redux/-/redux-5.0.1.tgz", "integrity": "sha512-M9/ELqF6fy8FwmkpnF0S3YKOqMyoWJ4+CS5Efg2ct3oY9daQvd/Pc71FpGZsVsbl3Cpb+IIcjBDUnnyBdQbq4w==", - "license": "MIT" + "license": "MIT", + "peer": true }, "node_modules/redux-thunk": { "version": "3.1.0", @@ -8870,6 +8866,7 @@ "integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==", "dev": true, "license": "MIT", + "peer": true, "engines": { "node": ">=12" }, @@ -9587,6 +9584,7 @@ "integrity": "sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ==", "dev": true, "license": "MIT", + "peer": true, "funding": { "url": "https://github.com/sponsors/colinhacks" } diff --git a/public/i18n/literals/fa.json b/public/i18n/literals/fa.json index 0f28367f..bfd2af60 100644 --- a/public/i18n/literals/fa.json +++ b/public/i18n/literals/fa.json @@ -1389,5 +1389,21 @@ "⚠️ Risk Notice: This provider uses a subscription/OAuth session not officially licensed for proxy/router use. Account may be restricted or banned. Use at your own risk.": "⚠️ اطلاعیه ریسک: این ارائه‌دهنده از اشتراک/جلسه OAuth استفاده می‌کند که به طور رسمی برای استفاده پروکسی/روتر مجوز ندارد. حساب ممکن است محدود یا مسدود شود. با مسئولیت خود استفاده کنید.", "✓ Confirm Add": "✓ تأیید افزودن", "📝 Configure providers in dashboard or use environment variables": "📝 ارائه‌دهندگان را در داشبورد پیکربندی کنید یا از متغیرهای محیطی استفاده کنید", - "🔐 OAuth required. Add now and authenticate after Apply; tool list will be discovered after first connect.": "🔐 نیاز به OAuth. اکنون اضافه کنید و پس از اعمال، احراز هویت کنید؛ لیست ابزارها پس از اولین اتصال کشف می‌شود." + "🔐 OAuth required. Add now and authenticate after Apply; tool list will be discovered after first connect.": "🔐 نیاز به OAuth. اکنون اضافه کنید و پس از اعمال، احراز هویت کنید؛ لیست ابزارها پس از اولین اتصال کشف می‌شود.", + "Combo & Vision Adapter":"آداپتور بینایی و ترکیبی", + "Vision Adapter": "آداپتور بینایی", + "Skills": "مهارت‌ها", + "Cached Cost":"هزینه‌ی کش‌شده", + "Compress prompts and outputs to save tokens":"فشرده‌سازی پرامپت‌ها و خروجی‌ها برای صرفه‌جویی در توکن‌ها", + "Manage your web providers": "مدیریت ارائه‌دهندگان خدمات وب خود را انجام دهید", + "providers":"ارائه‌دهندگان", + "combos": "ترکیبی", + "Configure enterprise Single Sign-On (SSO) for dashboard access using SAML 2.0 or OIDC.":"پیکربندی ورود تک‌نشانه سازمانی (SSO) برای دسترسی به داشبورد با استفاده از SAML 2.0 یا OIDC.", + "Optional SSO via Okta, Entra ID, Keycloak, or OIDC":"SSO اختیاری از طریق Okta، Entra ID، Keycloak یا OIDC", + "Single Sign-On (SSO)":"ورود یک‌باره (SSO)", + "SSO Protocol":"پروتکل SSO", + "Keep legacy password login.":"حفظ ورود با نام کاربری و رمز عبور قدیمی", + "Require SSO for dashboard access.":"برای دسترسی به داشبورد نیاز به SSO دارد.", + "Allow password or SSO login.":"اجازه ورود با رمز عبور یا ورود یک‌بار مصرف (SSO)", + "Save OIDC settings":"ذخیره تنظیمات OIDC" } diff --git a/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js b/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js index 209a6d33..82211e48 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js +++ b/src/app/(dashboard)/dashboard/cli-tools/[toolId]/ToolDetailClient.js @@ -8,7 +8,7 @@ import { getModelsByProviderId, PROVIDER_ID_TO_ALIAS } from "@/shared/constants/ import { ClaudeToolCard, CodexToolCard, DroidToolCard, OpenClawToolCard, HermesToolCard, DefaultToolCard, OpenCodeToolCard, CoworkToolCard, - CopilotToolCard, ClineToolCard, KiloToolCard, DeepSeekTuiToolCard, + ClineToolCard, KiloToolCard, DeepSeekTuiToolCard, JcodeToolCard, GrokBuildToolCard, } from "../components"; @@ -156,8 +156,6 @@ export default function ToolDetailClient({ toolId, machineId }) { return ; case "hermes": return ; - case "copilot": - return ; case "cline": return ; case "kilo": diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js b/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js index fad05f9e..f49d8a38 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js +++ b/src/app/(dashboard)/dashboard/cli-tools/components/ClaudeToolCard.js @@ -7,19 +7,25 @@ import BaseUrlSelect from "./BaseUrlSelect"; import { rememberEndpoint } from "./cliEndpointPresets"; import ApiKeySelect from "./ApiKeySelect"; import { matchKnownEndpoint } from "./cliEndpointMatch"; +import { stripModelContextMarker } from "open-sse/utils/modelMarkers.js"; const CLOUD_URL = process.env.NEXT_PUBLIC_CLOUD_URL; -// Context window presets. UI shows the round number; the value written is nudged -// down 2K to stay safely under the upstream hard cap. +// Auto-compact window presets (CLAUDE_CODE_AUTO_COMPACT_WINDOW, valid 100K–1M). +// UI shows the round number; the value written is nudged down 2K to stay safely +// under the upstream hard cap. const CONTEXT_OPTIONS = [ { label: "Default", value: "" }, { label: "200K", value: "198000" }, { label: "300K", value: "298000" }, { label: "500K", value: "498000" }, - { label: "1M", value: "998000" }, + { label: "700K", value: "698000" }, ]; +// Claude Code assumes a model's window is 200K unless the name carries the `[1m]` +// marker, which is why the 1M auto-compact preset only takes effect once the +// marker is applied. + export default function ClaudeToolCard({ tool, isExpanded, @@ -51,9 +57,28 @@ export default function ClaudeToolCard({ const [customBaseUrl, setCustomBaseUrl] = useState(""); const [ccFilterNaming, setCcFilterNaming] = useState(false); const [exaMcpEnabled, setExaMcpEnabled] = useState(false); - const [maxContextTokens, setMaxContextTokens] = useState(""); + const [autoCompactWindow, setAutoCompactWindow] = useState(""); + const [oneMContext, setOneMContext] = useState(false); const hasInitializedModels = useRef(false); + // Claude Code only string-matches the marker against the model name, so it + // applies to any id — the user decides which models are worth declaring as 1M. + // Stripping first keeps repeated toggles from stacking `[1m][1m]`. + const withContextMarker = (value, enabled) => { + const { model } = stripModelContextMarker(value); + return enabled ? `${model}[1m]` : model; + }; + + // Rewrite the mappings in place on toggle, so the inputs show what will be + // written without waiting for Apply. + const handleOneMContextToggle = (enabled) => { + setOneMContext(enabled); + tool.defaultModels.forEach((model) => { + const current = modelMappings[model.alias]; + if (current) onModelMappingChange(model.alias, withContextMarker(current, enabled)); + }); + }; + const currentBaseUrl = claudeStatus?.settings?.env?.ANTHROPIC_BASE_URL || ""; const getConfigStatus = () => { @@ -80,9 +105,15 @@ export default function ClaudeToolCard({ }, [initialStatus]); useEffect(() => { - const v = claudeStatus?.settings?.env?.CLAUDE_CODE_MAX_CONTEXT_TOKENS; - setMaxContextTokens(v || ""); - }, [claudeStatus?.settings?.env?.CLAUDE_CODE_MAX_CONTEXT_TOKENS]); + const v = claudeStatus?.settings?.env?.CLAUDE_CODE_AUTO_COMPACT_WINDOW; + setAutoCompactWindow(v || ""); + }, [claudeStatus?.settings?.env?.CLAUDE_CODE_AUTO_COMPACT_WINDOW]); + + useEffect(() => { + const env = claudeStatus?.settings?.env; + if (!env) return; + setOneMContext(tool.defaultModels.some((model) => env[model.envKey]?.endsWith("[1m]"))); + }, [claudeStatus?.settings?.env, tool.defaultModels]); useEffect(() => { if (isExpanded) { @@ -124,6 +155,8 @@ export default function ClaudeToolCard({ tool.defaultModels.forEach((model) => { if (model.envKey) { + // Kept verbatim (marker included) so the input matches what is on disk; + // withContextMarker strips before appending, so re-applying cannot double it. const value = env[model.envKey] || model.defaultValue || ""; // Only sync initial values from file once if (value) { @@ -180,15 +213,17 @@ export default function ClaudeToolCard({ tool.defaultModels.forEach((model) => { const targetModel = modelMappings[model.alias]; + // Written verbatim — the input may hold a marker typed by hand, and the + // toggle already decided the marker when it was flipped. if (targetModel && model.envKey) env[model.envKey] = targetModel; }); - if (maxContextTokens) { - env.CLAUDE_CODE_MAX_CONTEXT_TOKENS = maxContextTokens; + if (autoCompactWindow) { + env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = autoCompactWindow; } const res = await fetch("/api/cli-tools/claude-settings", { method: "POST", headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ env, exaMcpEnabled, maxContextTokens }), + body: JSON.stringify({ env, exaMcpEnabled, autoCompactWindow }), }); const data = await res.json(); if (res.ok) { @@ -217,7 +252,8 @@ export default function ClaudeToolCard({ tool.defaultModels.forEach((model) => onModelMappingChange(model.alias, model.defaultValue || "")); setSelectedApiKey(""); setExaMcpEnabled(false); - setMaxContextTokens(""); + setAutoCompactWindow(""); + setOneMContext(false); } else { setMessage({ type: "error", text: data.error || "Failed to reset settings" }); } @@ -247,8 +283,8 @@ export default function ClaudeToolCard({ const targetModel = modelMappings[model.alias]; if (targetModel && model.envKey) env[model.envKey] = targetModel; }); - if (maxContextTokens) { - env.CLAUDE_CODE_MAX_CONTEXT_TOKENS = maxContextTokens; + if (autoCompactWindow) { + env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = autoCompactWindow; } return [ @@ -374,17 +410,30 @@ export default function ClaudeToolCard({ ))} - {/* Context Window */} + {/* Auto-compact window */}
- Context window + Auto-compact arrow_forward - setAutoCompactWindow(e.target.value)} className="w-full min-w-0 px-2 py-2 bg-surface rounded border border-border text-xs focus:outline-none focus:ring-1 focus:ring-primary/50 sm:py-1.5"> {CONTEXT_OPTIONS.map((opt) => ( ))}
+ {/* 1M context */} +
+ 1M context + arrow_forward + +
+ {/* CC Filter Naming */}
Filter naming diff --git a/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js b/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js index 6e944d79..83d6c15b 100644 --- a/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js +++ b/src/app/(dashboard)/dashboard/cli-tools/components/ToolSummaryCard.js @@ -5,7 +5,8 @@ import Image from "next/image"; import { Card } from "@/shared/components"; // Derive simple connected/configured/not-installed status from API payload -function getStatus(status) { +function getStatus(status, tool) { + if (tool?.configType === "guide") return { label: "Guide", cls: "bg-blue-500/10 text-blue-600 dark:text-blue-400" }; if (!status) return { label: "Unknown", cls: "bg-gray-500/10 text-gray-500" }; if (!status.installed) return { label: "Not installed", cls: "bg-gray-500/10 text-gray-500" }; if (status.has9Router) return { label: "Connected", cls: "bg-green-500/10 text-green-600 dark:text-green-400" }; @@ -13,7 +14,7 @@ function getStatus(status) { } export default function ToolSummaryCard({ toolId, tool, status }) { - const s = getStatus(status); + const s = getStatus(status, tool); return ( diff --git a/src/app/(dashboard)/dashboard/profile/page.js b/src/app/(dashboard)/dashboard/profile/page.js index 39d15686..3628eb4c 100644 --- a/src/app/(dashboard)/dashboard/profile/page.js +++ b/src/app/(dashboard)/dashboard/profile/page.js @@ -93,6 +93,12 @@ export default function ProfilePage() { const [proxyLoading, setProxyLoading] = useState(false); const [proxyTestLoading, setProxyTestLoading] = useState(false); + const [isRemoteHost, setIsRemoteHost] = useState(false); + useEffect(() => { + if (typeof window !== "undefined") + setIsRemoteHost(!["localhost", "127.0.0.1", "::1"].includes(window.location.hostname)); + }, []); + useEffect(() => { fetch("/api/settings") .then((res) => res.json()) @@ -1680,7 +1686,7 @@ export default function ProfilePage() { {/* App Info */}

{APP_CONFIG.name} v{APP_CONFIG.version}

-

Local Mode - All data stored on your machine

+

{isRemoteHost ? "Remote Mode" : "Local Mode - All data stored on your machine"}

0 + const models = (providerId === "cursor" || providerId === "zed") && liveModels.length > 0 ? liveModels : staticModels; const providerAlias = getProviderAlias(providerId); @@ -521,11 +530,13 @@ export default function ProviderDetailPage() { }); }, [fetchConnections, fetchAliases, fetchCustomModels, fetchDisabledModels]); - // Cursor's model availability is account-specific and changes frequently. - // Load the active account's live catalog for the dashboard; the static - // registry remains the fallback while the request is pending or unavailable. + // Live per-connection catalogs (cursor, zed): the static registry carries + // no usable list, so resolve from the active connection. Fires only when + // the provider id or connection list changes — no polling, no loop. + // Cursor path is statement-identical to before; zed adds error surfacing. useEffect(() => { - if (providerId !== "cursor") { + const isLiveCatalog = providerId === "cursor" || providerId === "zed"; + if (!isLiveCatalog) { queueMicrotask(() => setLiveModels([])); return; } @@ -533,18 +544,32 @@ export default function ProviderDetailPage() { const connection = connections.find((item) => item.isActive !== false); if (!connection?.id) { queueMicrotask(() => setLiveModels([])); + if (providerId === "zed") setLiveModelsError(null); return; } let cancelled = false; + if (providerId === "zed") setLiveModelsError(null); fetch(`/api/providers/${connection.id}/models`, { cache: "no-store" }) - .then(async (res) => ({ ok: res.ok, data: await res.json() })) + .then(async (res) => ({ ok: res.ok, data: await res.json().catch(() => null) })) .then(({ ok, data }) => { - if (!cancelled && ok && Array.isArray(data.models) && data.models.length > 0) { + if (cancelled) return; + if (ok && Array.isArray(data?.models) && data.models.length > 0) { setLiveModels(data.models); + if (providerId === "zed" && data?.warning) setLiveModelsError(data.warning); + return; + } + if (providerId === "zed") { + setLiveModels([]); + setLiveModelsError(data?.warning || data?.error || "Zed returned no live models."); } }) - .catch(() => {}); + .catch(() => { + if (!cancelled && providerId === "zed") { + setLiveModels([]); + setLiveModelsError("Failed to reach the Zed model catalog."); + } + }); return () => { cancelled = true; }; }, [providerId, connections]); @@ -673,6 +698,53 @@ export default function ProviderDetailPage() { setImportingQoderModels(false); } }; + // Fetch the live Cline /models catalog and add every model not yet present. + // Cline and ClinePass share the same catalog endpoint (api.cline.bot/api/v1/models). + const handleImportClineModels = async () => { + if (importingClineModels) return; + const activeConnection = connections.find((conn) => conn.isActive !== false); + if (!activeConnection) { + alert(translate("Please add an active Cline connection first")); + return; + } + setImportingClineModels(true); + try { + const res = await fetch(`/api/providers/${activeConnection.id}/models`); + const data = await res.json(); + if (!res.ok) { + alert(data.error || translate("Failed to fetch models")); + return; + } + const models = data.models || []; + if (models.length === 0) { + alert(translate("No models returned")); + return; + } + let importedCount = 0; + for (const model of models) { + const modelId = model.id || model.name; + if (!modelId) continue; + const alreadyExists = customModels.some( + (entry) => entry.providerAlias === providerStorageAlias && entry.id === modelId && (entry.kind || entry.type || "llm") === "llm" + ) || Object.values(modelAliases).includes(`${providerStorageAlias}/${modelId}`); + if (alreadyExists) { + continue; + } + await handleAddCustomModel(modelId, "llm", providerStorageAlias); + importedCount += 1; + } + if (importedCount === 0) { + alert(translate("All models already exist, no new models added")); + } else { + alert(translate("Successfully added") + ` ${importedCount} ` + translate("models")); + } + } catch (error) { + console.log("Error importing Cline models:", error); + alert(translate("Error fetching models") + ": " + error.message); + } finally { + setImportingClineModels(false); + } + }; const handleRunOneByOneTest = async () => { if (oneByOneRunning || connections.length === 0) return; @@ -1275,6 +1347,20 @@ export default function ProviderDetailPage() { )} + {/* Import Cline /models catalog button — only show for cline and clinepass providers */} + {(providerId === "cline" || providerId === "clinepass") && connections.some((conn) => conn.isActive !== false) && ( + + )} + {/* Suggested models from provider API — show only models not yet added */} {suggestedModels.length > 0 && (() => { const addedFullModels = new Set([ @@ -1792,7 +1878,36 @@ export default function ProviderDetailPage() { })()}
{!!modelsTestError && ( -

{modelsTestError}

+
+ )} + {providerId === "zed" && !!liveModelsError && ( +

{liveModelsError}

)} {renderModelsSection()}
@@ -1829,6 +1944,13 @@ export default function ProviderDetailPage() { onClose={() => setShowOAuthModal(false)} /> )} + + {/* Xiaomi Desktop: auto-import local credentials modal */} + setShowXiaomiMimoModal(false)} + /> {providerId === "iflow" && ( {currentPageRows.map((quota) => { const isUnlimited = quota.unlimited === true; - const colors = getColorClasses(quota.remaining); + const isCreditBalance = quota.isCreditBalance === true; + const colors = isCreditBalance + ? { text: "text-blue-600 dark:text-blue-400", bg: "bg-blue-500", bgLight: "bg-blue-500/10", emoji: "💰" } + : getColorClasses(quota.remaining); const countdown = formatResetTime(quota.resetAt); const resetDisplay = formatResetTimeDisplay(quota.resetAt); // recurring defaults true: a missing flag means the quota @@ -186,7 +189,7 @@ export default function QuotaTable({ {/* Progress + used/total */}
- {!isUnlimited && ( + {!isUnlimited && !isCreditBalance && (
@@ -203,15 +206,19 @@ export default function QuotaTable({ title={ isUnlimited ? `${quota.used.toLocaleString()} used · Unlimited` + : isCreditBalance + ? `Credit balance: ${quota.total.toFixed(2)} ${quota.currency || ""}` : `${quota.used.toLocaleString()} / ${quota.total > 0 ? quota.total.toLocaleString() : "∞"}` } > {isUnlimited ? `${quota.used.toLocaleString()} used · Unlimited` + : isCreditBalance + ? `Credit: ${quota.total.toFixed(2)} ${quota.currency || ""}` : `${quota.used.toLocaleString()} / ${quota.total > 0 ? quota.total.toLocaleString() : "∞"}`} - - {isUnlimited ? "Unlimited" : `${quota.remaining}%`} + + {isUnlimited ? "Unlimited" : isCreditBalance ? "" : `${quota.remaining}%`}
diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js index 85fed92b..80678396 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/index.js @@ -41,7 +41,7 @@ import { } from "./utils"; import Card from "@/shared/components/Card"; import { ConfirmModal, EditConnectionModal } from "@/shared/components"; -import { USAGE_SUPPORTED_PROVIDERS } from "@/shared/constants/providers"; +import { USAGE_SUPPORTED_PROVIDERS, AI_PROVIDERS } from "@/shared/constants/providers"; import { useCopyToClipboard } from "@/shared/hooks/useCopyToClipboard"; // Maps the stored providerSpecificData.authMethod to a human label for Kiro. @@ -101,6 +101,10 @@ function getCodexResetCreditCount(quota) { return Number.isFinite(count) ? Math.max(0, count) : 0; } +function providerLabel(providerId) { + return AI_PROVIDERS[providerId]?.name || providerId; +} + function formatCreditDate(value) { if (!value) return "N/A"; const date = new Date(value); @@ -597,6 +601,17 @@ export default function ProviderLimits() { const providerVisibility = previous[provider] || {}; const hidden = new Set(providerVisibility.hidden || []); hidden.add(key); + if (provider === "antigravity") { + if (key === "gemini") { + for (const k of hidden) { + if (k.startsWith("gemini-") && !k.includes("image")) hidden.delete(k); + } + } else if (key === "claude") { + for (const k of hidden) { + if (k.startsWith("claude-")) hidden.delete(k); + } + } + } const next = { ...previous, [provider]: { @@ -615,6 +630,17 @@ export default function ProviderLimits() { const providerVisibility = previous[provider] || {}; const hidden = new Set(providerVisibility.hidden || []); hidden.delete(key); + if (provider === "antigravity") { + if (key === "gemini") { + for (const k of hidden) { + if (k.startsWith("gemini-") && !k.includes("image")) hidden.delete(k); + } + } else if (key === "claude") { + for (const k of hidden) { + if (k.startsWith("claude-")) hidden.delete(k); + } + } + } const next = { ...previous, [provider]: { @@ -746,7 +772,7 @@ export default function ProviderLimits() { }; const selectedProviderLabel = - providerFilter === "all" ? "All providers" : providerFilter; + providerFilter === "all" ? "All providers" : providerLabel(providerFilter); const hasEligibleConnections = totals.eligibleConnections > 0; const hasVisibleConnections = sortedConnections.length > 0; const emptyState = getConnectionsEmptyMessage( @@ -823,7 +849,7 @@ export default function ProviderLimits() { fallbackText={providerFilter.slice(0, 2).toUpperCase()} /> )} - + {selectedProviderLabel} @@ -884,8 +910,8 @@ export default function ProviderLimits() { className="size-6 rounded-md object-contain" fallbackText={provider.slice(0, 2).toUpperCase()} /> - - {provider} + + {providerLabel(provider)} {providerFilter === provider && ( @@ -1058,8 +1084,8 @@ export default function ProviderLimits() { />
-

- {conn.provider} +

+ {providerLabel(conn.provider)}

{getConnectionLabel(conn) ? (

diff --git a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js index f8d8b760..7e341eea 100644 --- a/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js +++ b/src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js @@ -359,21 +359,33 @@ export function getQuotaVisibilityKey(quota) { return String(quota.modelKey || quota.name || "").trim(); } -function getProviderHiddenQuotaSet(provider, quotaVisibility) { +/** + * Trim hidden quota keys to only those matching currently valid quotas. + * Stale or obsolete model keys are dropped. + */ +export function trimHiddenQuotaKeys(hidden = [], quotas = []) { + if (!Array.isArray(hidden) || hidden.length === 0) return []; + const validKeys = new Set(quotas.map(getQuotaVisibilityKey).filter(Boolean)); + return [...new Set(hidden.map((k) => String(k).trim()).filter((k) => validKeys.has(k)))]; +} + +function getProviderHiddenQuotaSet(provider, quotaVisibility, quotas = []) { const hidden = quotaVisibility?.[provider]?.hidden; - return new Set(Array.isArray(hidden) ? hidden.map(String) : []); + if (!Array.isArray(hidden) || hidden.length === 0) return new Set(); + const trimmed = quotas.length > 0 ? trimHiddenQuotaKeys(hidden, quotas) : hidden; + return new Set(trimmed.map(String)); } export function filterQuotasByVisibility(provider, quotas = [], quotaVisibility = {}) { if (!Array.isArray(quotas) || quotas.length === 0) return []; - const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility); + const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility, quotas); if (hidden.size === 0) return quotas; return quotas.filter((quota) => !hidden.has(getQuotaVisibilityKey(quota))); } export function getHiddenQuotaRows(provider, quotas = [], quotaVisibility = {}) { if (!Array.isArray(quotas) || quotas.length === 0) return []; - const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility); + const hidden = getProviderHiddenQuotaSet(provider, quotaVisibility, quotas); if (hidden.size === 0) return []; return quotas.filter((quota) => hidden.has(getQuotaVisibilityKey(quota))); } @@ -406,10 +418,68 @@ export function parseQuotaData(provider, data) { case "antigravity": if (data.quotas) { - Object.entries(data.quotas).forEach(([modelKey, quota]) => { + const entries = Object.entries(data.quotas); + const weeklyKeys = new Set(["gemini_weekly", "claude_gpt_weekly"]); + const geminiModels = entries.filter(([k]) => k.startsWith("gemini-") && !k.includes("image")); + const claudeModels = entries.filter(([k]) => k.startsWith("claude-")); + const imageModels = entries.filter(([k]) => k.includes("image")); + const weeklyModels = entries.filter(([k]) => weeklyKeys.has(k)); + const otherModels = entries.filter(([k]) => !k.startsWith("gemini-") && !k.startsWith("claude-") && !k.includes("image") && !weeklyKeys.has(k)); + + if (geminiModels.length > 0) { + const rep = geminiModels.reduce((min, cur) => + (cur[1].remainingPercentage ?? 100) < (min[1].remainingPercentage ?? 100) ? cur : min + )[1]; + normalizedQuotas.push({ + name: "Gemini (Flash / Pro)", + modelKey: "gemini", + used: rep.used || 0, + total: rep.total || 0, + resetAt: rep.resetAt || null, + remainingPercentage: rep.remainingPercentage, + }); + } + + if (claudeModels.length > 0) { + const rep = claudeModels.reduce((min, cur) => + (cur[1].remainingPercentage ?? 100) < (min[1].remainingPercentage ?? 100) ? cur : min + )[1]; + normalizedQuotas.push({ + name: "Claude (Sonnet / Opus)", + modelKey: "claude", + used: rep.used || 0, + total: rep.total || 0, + resetAt: rep.resetAt || null, + remainingPercentage: rep.remainingPercentage, + }); + } + + weeklyModels.forEach(([modelKey, quota]) => { normalizedQuotas.push({ name: quota.displayName || modelKey, - modelKey: modelKey, // Keep modelKey for sorting + modelKey, + used: quota.used || 0, + total: quota.total || 0, + resetAt: quota.resetAt || null, + remainingPercentage: quota.remainingPercentage, + }); + }); + + imageModels.forEach(([modelKey, quota]) => { + normalizedQuotas.push({ + name: quota.displayName || modelKey, + modelKey, + used: quota.used || 0, + total: quota.total || 0, + resetAt: quota.resetAt || null, + remainingPercentage: quota.remainingPercentage, + }); + }); + + otherModels.forEach(([modelKey, quota]) => { + normalizedQuotas.push({ + name: quota.displayName || modelKey, + modelKey, used: quota.used || 0, total: quota.total || 0, resetAt: quota.resetAt || null, @@ -496,6 +566,8 @@ export function parseQuotaData(provider, data) { name, used: quota.used || 0, total: quota.total || 0, + remaining: quota.remaining !== undefined ? quota.remaining : Math.max(0, (quota.total || 100) - (quota.used || 0)), + remainingPercentage: quota.remainingPercentage !== undefined ? quota.remainingPercentage : calculatePercentage(quota.used, quota.total), resetAt: quota.resetAt || null, }); }); @@ -580,6 +652,8 @@ export function parseQuotaData(provider, data) { total: quota.total || 0, resetAt: quota.resetAt || null, remainingPercentage: quota.remainingPercentage, + isCreditBalance: quota.isCreditBalance ?? true, + currency: quota.currency || (name.includes("(") ? name.slice(name.indexOf("(") + 1, name.indexOf(")")) : "USD"), }); }); } @@ -671,15 +745,31 @@ export function parseQuotaData(provider, data) { return []; } + if (provider?.toLowerCase() === "claude") { + const CLAUDE_QUOTA_ORDER = { + "session (5h)": 0, + "weekly (7d)": 1, + "weekly fable (7d)": 2, + "weekly opus (7d)": 3, + "weekly sonnet (7d)": 4, + }; + normalizedQuotas.sort((a, b) => (CLAUDE_QUOTA_ORDER[a.name] ?? 99) - (CLAUDE_QUOTA_ORDER[b.name] ?? 99)); + return normalizedQuotas; + } + // Sort quotas according to PROVIDER_MODELS order const modelOrder = getModelsByProviderId(provider); if (modelOrder.length > 0) { const orderMap = new Map(modelOrder.map((m, i) => [m.id, i])); normalizedQuotas.sort((a, b) => { - // Use modelKey for antigravity, otherwise use name - const keyA = a.modelKey || a.name; - const keyB = b.modelKey || b.name; + // Use modelKey for antigravity (mapped to family anchor), otherwise use name + let keyA = a.modelKey || a.name; + let keyB = b.modelKey || b.name; + if (keyA === "gemini") keyA = "gemini-3.8-flash-high"; + if (keyA === "claude") keyA = "claude-sonnet-4-6"; + if (keyB === "gemini") keyB = "gemini-3.8-flash-high"; + if (keyB === "claude") keyB = "claude-sonnet-4-6"; const orderA = orderMap.get(keyA) ?? 999; const orderB = orderMap.get(keyB) ?? 999; return orderA - orderB; diff --git a/src/app/api/cli-tools/all-statuses/route.js b/src/app/api/cli-tools/all-statuses/route.js index 5925e526..b4270163 100644 --- a/src/app/api/cli-tools/all-statuses/route.js +++ b/src/app/api/cli-tools/all-statuses/route.js @@ -8,7 +8,6 @@ import { GET as droidGet } from "../droid-settings/route"; import { GET as openclawGet } from "../openclaw-settings/route"; import { GET as hermesGet } from "../hermes-settings/route"; import { GET as coworkGet } from "../cowork-settings/route"; -import { GET as copilotGet } from "../copilot-settings/route"; import { GET as clineGet } from "../cline-settings/route"; import { GET as kiloGet } from "../kilo-settings/route"; import { GET as deepseekTuiGet } from "../deepseek-tui-settings/route"; @@ -24,7 +23,6 @@ const STATUS_GETTERS = { openclaw: openclawGet, hermes: hermesGet, cowork: coworkGet, - copilot: copilotGet, cline: clineGet, kilo: kiloGet, "deepseek-tui": deepseekTuiGet, diff --git a/src/app/api/cli-tools/claude-settings/route.js b/src/app/api/cli-tools/claude-settings/route.js index 76ba9232..c7087577 100644 --- a/src/app/api/cli-tools/claude-settings/route.js +++ b/src/app/api/cli-tools/claude-settings/route.js @@ -123,7 +123,7 @@ export async function GET() { // POST - Backup old fields and write new settings export async function POST(request) { try { - const { env, exaMcpEnabled, maxContextTokens } = await request.json(); + const { env, exaMcpEnabled, autoCompactWindow } = await request.json(); if (!env || typeof env !== "object") { return NextResponse.json( @@ -166,12 +166,13 @@ export async function POST(request) { }, }; - // CLAUDE_CODE_MAX_CONTEXT_TOKENS — only set when a concrete value is chosen; - // "Default" removes the key so Claude Code falls back to the model's window. - if (maxContextTokens) { - newSettings.env.CLAUDE_CODE_MAX_CONTEXT_TOKENS = String(maxContextTokens); + // CLAUDE_CODE_AUTO_COMPACT_WINDOW — the token threshold that triggers + // auto-compact. Only set when a concrete value is chosen; "Default" removes + // the key so Claude Code derives the window from the model. + if (autoCompactWindow) { + newSettings.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = String(autoCompactWindow); } else { - delete newSettings.env.CLAUDE_CODE_MAX_CONTEXT_TOKENS; + delete newSettings.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW; } // Write new settings @@ -203,7 +204,7 @@ const RESET_ENV_KEYS = [ "ANTHROPIC_DEFAULT_SONNET_MODEL", "ANTHROPIC_DEFAULT_HAIKU_MODEL", "API_TIMEOUT_MS", - "CLAUDE_CODE_MAX_CONTEXT_TOKENS", + "CLAUDE_CODE_AUTO_COMPACT_WINDOW", ]; // DELETE - Reset settings (remove env fields) diff --git a/src/app/api/cli-tools/cowork-mcp-tools/route.js b/src/app/api/cli-tools/cowork-mcp-tools/route.js index 5cc3d3a1..539dcd70 100644 --- a/src/app/api/cli-tools/cowork-mcp-tools/route.js +++ b/src/app/api/cli-tools/cowork-mcp-tools/route.js @@ -1,6 +1,8 @@ "use server"; import { NextResponse } from "next/server"; +import { assertPublicUrl } from "@/shared/utils/ssrfGuard.js"; +import { isLocalRequest } from "@/dashboardGuard"; const TIMEOUT_MS = 8000; @@ -87,6 +89,14 @@ export async function POST(request) { if (!url || typeof url !== "string") { return NextResponse.json({ error: "url required" }, { status: 400 }); } + // SSRF guard for remote callers; local host keeps self-hosted MCP servers. + if (!isLocalRequest(request)) { + try { + assertPublicUrl(url); + } catch { + return NextResponse.json({ error: "URL not allowed" }, { status: 400 }); + } + } const result = await probeMcp(url); return NextResponse.json(result); } catch (e) { diff --git a/src/app/api/models/test/ping.js b/src/app/api/models/test/ping.js index 24119eb2..a5c882d5 100644 --- a/src/app/api/models/test/ping.js +++ b/src/app/api/models/test/ping.js @@ -1,4 +1,6 @@ import { getApiKeys } from "@/lib/localDb"; +import { resolveProviderId } from "@/shared/constants/providers.js"; +import { unwrapClineEnvelope } from "open-sse/shared/clineEnvelope.js"; import { UPDATER_CONFIG } from "@/shared/constants/config"; import { getConsistentMachineId } from "@/shared/utils/machineId"; @@ -151,9 +153,14 @@ export async function pingModelByKind(model, kind, baseUrl = `http://127.0.0.1:$ let parsed = null; try { parsed = rawText ? JSON.parse(rawText) : null; } catch {} + // Unwrap before the choices checks below. No-op for providers that do not + // opt in via transport.quirks.clineEnvelope. + const providerId = resolveProviderId(String(model).split("/")[0]); + parsed = unwrapClineEnvelope(parsed, providerId); + if (!res.ok) { const detail = parsed?.error?.message || parsed?.msg || parsed?.message || parsed?.error || rawText; - return { ok: false, latencyMs, error: `HTTP ${res.status}${detail ? `: ${String(detail).slice(0, 240)}` : ""}`, status: res.status }; + return { ok: false, latencyMs, error: `HTTP ${res.status}${detail ? `: ${String(detail).slice(0, 500)}` : ""}`, status: res.status }; } const providerStatus = parsed?.status; diff --git a/src/app/api/oauth/[provider]/[action]/route.js b/src/app/api/oauth/[provider]/[action]/route.js index 5b86f013..8c56fb60 100644 --- a/src/app/api/oauth/[provider]/[action]/route.js +++ b/src/app/api/oauth/[provider]/[action]/route.js @@ -1,3 +1,4 @@ +import crypto from "crypto"; import { NextResponse } from "next/server"; import { getProvider, @@ -7,6 +8,7 @@ import { pollForToken } from "@/lib/oauth/providers"; import { createProviderConnection } from "@/models"; +import { readDesktopPassToken } from "open-sse/shared/mimoAccount.js"; import { startCodexProxy, stopCodexProxy, @@ -33,6 +35,11 @@ import { registerZedSession, getZedSessionStatus, clearZedSession, + startXiaomiMimoProxy, + stopXiaomiMimoProxy, + registerXiaomiMimoSession, + getXiaomiMimoSessionStatus, + clearXiaomiMimoSession, } from "@/lib/oauth/utils/server"; import { detectIdeInstalled } from "@/lib/oauth/utils/ideDetect"; import { ZED_HOSTED_CONFIG } from "@/lib/oauth/constants/oauth"; @@ -89,6 +96,32 @@ export async function GET(request, { params }) { const { searchParams } = new URL(request.url); if (action === "authorize") { + // Xiaomi Desktop: custom ECDH flow — generate keypair, start proxy, return authorize URL + if (provider === "xiaomi-mimo") { + const { generateKeyPair, buildAuthorizeUrl, getKeyName } = await import("@/lib/oauth/providers/xiaomi-mimo"); + const { publicKey, privateKeyDer } = generateKeyPair(); + const state = searchParams.get("state") || crypto.randomUUID(); + + // Start the callback proxy (or reuse if already running) + const proxyResult = await startXiaomiMimoProxy(); + if (!proxyResult.success) { + return NextResponse.json({ error: `Failed to start callback server: ${proxyResult.reason}` }, { status: 500 }); + } + + // Register the session with the private key for decryption + registerXiaomiMimoSession({ state, privateKeyDer }); + + const redirectUri = proxyResult.callbackUrl; + const authorizeUrl = buildAuthorizeUrl(publicKey, redirectUri, getKeyName()); + + return NextResponse.json({ + state, + authorizeUrl, + redirectUri, + port: proxyResult.port, + }); + } + const redirectUri = searchParams.get("redirect_uri") || "http://localhost:8080/callback"; // Collect provider-specific meta params (e.g. gitlab passes baseUrl, clientId, clientSecret) const reservedParams = new Set(["redirect_uri"]); @@ -120,6 +153,10 @@ export async function GET(request, { params }) { const result = await startZedProxy(searchParams.get("native_app_port") || ZED_HOSTED_CONFIG.defaultNativeAppPort); return NextResponse.json(result); } + if (provider === "xiaomi-mimo") { + const result = await startXiaomiMimoProxy(); + return NextResponse.json(result); + } if (!["codex", "xai"].includes(provider)) { return NextResponse.json({ error: "Proxy only supported for codex/xai/trae/windsurf/zed" }, { status: 400 }); } @@ -153,10 +190,21 @@ export async function GET(request, { params }) { else if (provider === "zed") session = getZedSessionStatus(state); else if (provider === "xai") session = getXaiSessionStatus(state); else if (provider === "codex") session = getCodexSessionStatus(state); - else return NextResponse.json({ error: "Poll only supported for codex/xai/trae/windsurf/zed" }, { status: 400 }); + else if (provider === "xiaomi-mimo") session = getXiaomiMimoSessionStatus(state); + else return NextResponse.json({ error: "Poll only supported for codex/xai/trae/windsurf/zed/xiaomi-mimo" }, { status: 400 }); if (!session) return NextResponse.json({ status: "unknown" }); if (session.status === "done" || session.status === "error") { const payload = { ...session }; + if (provider === "xiaomi-mimo") { + // Unlike the others this does not auto-exchange server-side, so a + // finished session must survive until the client POSTs /exchange — + // that call clears it. A failed one is cleared here instead. + if (session.status === "error") { + clearXiaomiMimoSession(state); + stopXiaomiMimoProxy(); + } + return NextResponse.json(payload); + } if (provider === "trae") clearTraeSession(state); else if (provider === "windsurf") clearWindsurfSession(state); else if (provider === "zed") clearZedSession(state); @@ -173,7 +221,8 @@ export async function GET(request, { params }) { else if (provider === "zed") stopZedProxy(); else if (provider === "xai") stopXaiProxy(); else if (provider === "codex") stopCodexProxy(); - else return NextResponse.json({ error: "Proxy only supported for codex/xai/trae/windsurf/zed" }, { status: 400 }); + else if (provider === "xiaomi-mimo") stopXiaomiMimoProxy(); + else return NextResponse.json({ error: "Proxy only supported for codex/xai/trae/windsurf/zed/xiaomi-mimo" }, { status: 400 }); return NextResponse.json({ success: true }); } @@ -261,13 +310,75 @@ export async function POST(request, { params }) { let ok = false; if (provider === "trae") ok = registerTraeSession({ state }); else if (provider === "windsurf") ok = registerWindsurfSession({ state }); - else if (provider === "zed") ok = registerZedSession({ state, codeVerifier: body?.codeVerifier }); + else if (provider === "zed") ok = registerZedSession({ state, codeVerifier: body?.codeVerifier, systemId: body?.systemId }); else return NextResponse.json({ error: "register-session only supported for trae/windsurf/zed" }, { status: 400 }); return NextResponse.json({ success: ok }); } if (action === "exchange") { - const { code, redirectUri, codeVerifier, state, meta } = body; + const { code, redirectUri, codeVerifier, state, meta, systemId } = body; + + // Xiaomi MiMo: no token exchange needed — the callback already decrypted the sk. + // Just read the session result and create the connection. + if (provider === "xiaomi-mimo") { + if (!state) { + return NextResponse.json({ error: "Missing state" }, { status: 400 }); + } + const session = getXiaomiMimoSessionStatus(state); + if (!session || session.status !== "done" || !session.result) { + return NextResponse.json( + { error: session?.error || "OAuth session not completed. Please restart the login flow." }, + { status: 400 }, + ); + } + const { uid, accessToken, baseUrl } = session.result; + + // Desktop-exclusive Preview models authenticate with the account-session + // passToken, which only lives in MiMo Desktop's cookie store — attach it + // to the connection so those models work right after OAuth. + let passToken = null; + try { + passToken = await readDesktopPassToken(); + } catch { + // Desktop not installed / cookie DB locked — preview models stay unavailable. + } + + try { + const connection = await createProviderConnection({ + provider: "xiaomi-mimo", + authType: "oauth", + accessToken, + refreshToken: null, + expiresAt: new Date(Date.now() + 365 * 24 * 60 * 60 * 1000).toISOString(), + email: uid ? `${uid}@xiaomi` : null, + displayName: uid ? `Xiaomi ${uid}` : "Xiaomi MiMo", + providerSpecificData: { + uid: uid || null, + baseUrl: baseUrl || "https://api.xiaomimimo.com/v1", + authMethod: "oauth", + mimoPassToken: passToken?.passToken || null, + mimoUserId: passToken?.userId || null, + mimoCUserId: passToken?.cUserId || null, + }, + testStatus: "active", + }); + clearXiaomiMimoSession(state); + stopXiaomiMimoProxy(); + return NextResponse.json({ + success: true, + connection: { + id: connection.id, + provider: connection.provider, + email: connection.email, + displayName: connection.displayName, + }, + }); + } catch (err) { + clearXiaomiMimoSession(state); + stopXiaomiMimoProxy(); + return NextResponse.json({ error: err.message }, { status: 500 }); + } + } // Trae/Windsurf: code is either a raw callback URL or a pasted token. // exchangeTokens() handles both paths; no PKCE, skip codex JWT extraction. @@ -349,8 +460,13 @@ export async function POST(request, { params }) { return NextResponse.json({ error: "Missing required fields" }, { status: 400 }); } - // Exchange code for tokens (meta carries provider-specific params, e.g. gitlab clientId/baseUrl) - const tokenData = await exchangeTokens(provider, code, redirectUri, codeVerifier, state, meta); + // Exchange code for tokens (meta carries provider-specific params, e.g. gitlab clientId/baseUrl). + // systemId (Zed) is merged into meta so the login attempt's own id is + // used instead of a freshly prepared one. Ignored by other providers. + const tokenData = await exchangeTokens(provider, code, redirectUri, codeVerifier, state, { + ...(meta || {}), + ...(systemId ? { systemId } : {}), + }); // Save to database const connection = await createProviderConnection({ diff --git a/src/app/api/oauth/xiaomi-mimo/api-key/route.js b/src/app/api/oauth/xiaomi-mimo/api-key/route.js new file mode 100644 index 00000000..d8ceee97 --- /dev/null +++ b/src/app/api/oauth/xiaomi-mimo/api-key/route.js @@ -0,0 +1,136 @@ +import { NextResponse } from "next/server"; +import { createProviderConnection } from "@/models"; + +/** + * POST /api/oauth/xiaomi-mimo/api-key + * Import a Xiaomi MiMo API key manually (or from auto-import). + * The key is validated against the models endpoint, then stored. + * + * Body: { apiKey, uid?, baseUrl? } + */ +export async function POST(request) { + try { + const { apiKey, uid, baseUrl, mimoPassToken, mimoUserId, mimoCUserId } = await request.json(); + + if (!apiKey || typeof apiKey !== "string" || !apiKey.trim()) { + return NextResponse.json( + { error: "API key is required" }, + { status: 400 }, + ); + } + + const key = apiKey.trim(); + if (!key.startsWith("sk-")) { + return NextResponse.json( + { error: "Invalid key format — expected sk- prefix" }, + { status: 400 }, + ); + } + + const effectiveBaseUrl = (baseUrl || "https://api.xiaomimimo.com/v1").replace(/\/+$/, ""); + + // Validate the key against the models endpoint + let validated = false; + let modelCount = 0; + try { + const resp = await fetch(`${effectiveBaseUrl}/models`, { + method: "GET", + headers: { + Authorization: `Bearer ${key}`, + "X-Mimo-Source": "mimocode-cli", + }, + signal: AbortSignal.timeout(10000), + }); + if (resp.ok) { + const data = await resp.json(); + modelCount = Array.isArray(data?.data) ? data.data.length : 0; + validated = true; + } + } catch { + // Network error — still allow import (key may be valid but network blocked) + } + + if (!validated) { + // Soft-fail: store the key but mark as untested + console.log("[xiaomi-mimo] key validation failed, storing as untested"); + } + + // Dedup: if a connection with the same uid or same key already exists, update it + const { getProviderConnections, updateProviderConnection } = await import("@/models"); + const existing = (await getProviderConnections()).find( + (c) => c.provider === "xiaomi-mimo" && ( + (uid && c.email === `${uid}@xiaomi`) || + c.accessToken === key + ), + ); + if (existing) { + const updated = await updateProviderConnection(existing.id, { + accessToken: key, + providerSpecificData: { + ...existing.providerSpecificData, + uid: uid || existing.providerSpecificData?.uid || null, + baseUrl: effectiveBaseUrl, + // Per-account session credential — enables multi-account rotation. + mimoPassToken: mimoPassToken || existing.providerSpecificData?.mimoPassToken || null, + mimoUserId: mimoUserId || existing.providerSpecificData?.mimoUserId || null, + mimoCUserId: mimoCUserId || existing.providerSpecificData?.mimoCUserId || null, + modelCount, + }, + testStatus: validated ? "active" : existing.testStatus, + }); + return NextResponse.json({ + success: true, + validated, + modelCount, + updated: true, + connection: { + id: existing.id, + provider: existing.provider, + email: existing.email, + displayName: existing.displayName, + }, + }); + } + + const connection = await createProviderConnection({ + provider: "xiaomi-mimo", + authType: "api_key", + accessToken: key, + refreshToken: null, + // API keys don't expire on a fixed schedule; use a long horizon + expiresAt: new Date(Date.now() + 365 * 24 * 60 * 60 * 1000).toISOString(), + email: uid ? `${uid}@xiaomi` : null, + displayName: uid ? `Xiaomi ${uid}` : "Xiaomi MiMo", + providerSpecificData: { + uid: uid || null, + baseUrl: effectiveBaseUrl, + authMethod: "api_key", + provider: "API Key", + modelCount, + // Per-account session credential — enables multi-account rotation. + mimoPassToken: mimoPassToken || null, + mimoUserId: mimoUserId || null, + mimoCUserId: mimoCUserId || null, + }, + testStatus: validated ? "active" : "untested", + }); + + return NextResponse.json({ + success: true, + validated, + modelCount, + connection: { + id: connection.id, + provider: connection.provider, + email: connection.email, + displayName: connection.displayName, + }, + }); + } catch (error) { + console.log("Xiaomi MiMo API key import error:", error); + return NextResponse.json( + { error: "API key import failed" }, + { status: 500 }, + ); + } +} diff --git a/src/app/api/oauth/xiaomi-mimo/auto-import/route.js b/src/app/api/oauth/xiaomi-mimo/auto-import/route.js new file mode 100644 index 00000000..0508c222 --- /dev/null +++ b/src/app/api/oauth/xiaomi-mimo/auto-import/route.js @@ -0,0 +1,140 @@ +import { NextResponse } from "next/server"; +import { readFile, access, constants } from "fs/promises"; +import { homedir } from "os"; +import { join } from "path"; +import { readDesktopPassToken } from "open-sse/shared/mimoAccount.js"; + +/** + * GET /api/oauth/xiaomi-mimo/auto-import + * Auto-detect Xiaomi MiMo credentials from local auth.json. + * + * Sources (in priority order): + * 1. ~/.local/share/mimocode/auth.json → xiaomi field + * 2. %APPDATA%/Xiaomi MiMo/... → (future: Desktop keychain) + * + * auth.json shape: + * { + * "xiaomi": { + * "type": "api", + * "key": "sk-xxxx", + * "metadata": { "uid": "...", "base_url": "https://api.xiaomimimo.com/v1" } + * } + * } + */ + +function getCandidatePaths() { + const home = homedir(); + const paths = []; + + // MiMoCode / MiMo Desktop shared data dir (cross-platform XDG) + paths.push(join(home, ".local", "share", "mimocode", "auth.json")); + + // Windows: also check USERPROFILE-based XDG + if (process.platform === "win32") { + const appData = process.env.APPDATA || join(home, "AppData", "Roaming"); + // Desktop's own storage (may have separate credentials in the future) + paths.push(join(appData, "Xiaomi MiMo", "auth.json")); + } + + // macOS + if (process.platform === "darwin") { + paths.push( + join(home, "Library", "Application Support", "mimocode", "auth.json"), + ); + } + + return paths; +} + +/** + * GET /api/oauth/xiaomi-mimo/auto-import + */ +export async function GET() { + try { + const candidates = getCandidatePaths(); + + let authPath = null; + for (const candidate of candidates) { + try { + await access(candidate, constants.R_OK); + authPath = candidate; + break; + } catch { + // Try next candidate + } + } + + if (!authPath) { + return NextResponse.json({ + found: false, + error: `Xiaomi MiMo Desktop auth file not found. Checked:\n${candidates.join("\n")}\n\nMake sure Xiaomi MiMo Desktop is installed and you are signed in.`, + }); + } + + const raw = await readFile(authPath, "utf-8"); + let auth; + try { + auth = JSON.parse(raw); + } catch { + return NextResponse.json({ + found: false, + error: "auth.json is not valid JSON. Please sign in to Xiaomi MiMo Desktop again.", + }); + } + + const xiaomi = auth?.xiaomi; + if (!xiaomi || !xiaomi.key) { + return NextResponse.json({ + found: false, + error: "No Xiaomi credentials found in auth.json. Please sign in to Xiaomi MiMo Desktop.", + }); + } + + // Validate key format + const key = String(xiaomi.key).trim(); + if (!key.startsWith("sk-")) { + return NextResponse.json({ + found: false, + error: "Xiaomi key does not appear to be a valid API key (expected sk- prefix).", + }); + } + + const metadata = xiaomi.metadata || {}; + const uid = metadata.uid || null; + const baseUrl = metadata.base_url || "https://api.xiaomimimo.com/v1"; + + // Account-session passToken from Desktop's cookie store. Persisting it per + // connection is what lets multiple Xiaomi accounts rotate independently. + // (null while Desktop is running — its cookie DB is exclusively locked.) + let mimoPassToken = null; + let mimoUserId = null; + let mimoCUserId = null; + try { + const pt = await readDesktopPassToken(); + if (pt) { + mimoPassToken = pt.passToken; + mimoUserId = pt.userId; + mimoCUserId = pt.cUserId; + } + } catch (e) { + console.log("[xiaomi-mimo] passToken read failed (non-fatal):", e.message); + } + + return NextResponse.json({ + found: true, + apiKey: key, + uid, + baseUrl, + source: authPath, + mimoPassToken, + mimoUserId, + mimoCUserId, + }); + } catch (error) { + console.log("Xiaomi MiMo auto-import error:", error); + return NextResponse.json( + { found: false, error: error.message }, + { status: 500 }, + ); + } +} diff --git a/src/app/api/providers/[id]/models/route.js b/src/app/api/providers/[id]/models/route.js index 8a422b58..e0096e7e 100644 --- a/src/app/api/providers/[id]/models/route.js +++ b/src/app/api/providers/[id]/models/route.js @@ -1,7 +1,7 @@ import { NextResponse } from "next/server"; import { getProviderConnectionById } from "@/models"; import { isOpenAICompatibleProvider, isAnthropicCompatibleProvider } from "@/shared/constants/providers"; -import { GEMINI_CONFIG } from "@/lib/oauth/constants/oauth"; +import { GEMINI_CONFIG, ZED_HOSTED_CONFIG } from "@/lib/oauth/constants/oauth"; import { refreshGoogleToken, refreshCodexToken, updateProviderCredentials } from "@/sse/services/tokenRefresh"; import { resolveOllamaLocalHost } from "open-sse/config/providers.js"; import { getModelsByProviderId } from "open-sse/config/providerModels.js"; @@ -11,6 +11,8 @@ import { resolveQoderModels } from "open-sse/services/qoderModels.js"; import { resolveGrokCliModels } from "open-sse/services/grokCliModels.js"; import { resolveConnectionProxyConfig } from "@/lib/network/connectionProxy"; import { resolveCursorModels } from "open-sse/services/cursorModels.js"; +import { resolveZedModels } from "open-sse/shared/zedAuth.js"; +import { resolveClineModels, resolveClinepassModels } from "open-sse/services/clinepassModels.js"; const GEMINI_CLI_MODELS_URL = "https://cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels"; @@ -286,6 +288,75 @@ const PROVIDER_MODELS_CONFIG = { }; }, }, + // Zed has no static catalog by design (live /models only) — same cursor + // direct pattern: resolve with the connection's own credentials (never + // exposed to the browser), return rich metadata, drop disabled entries. + // Empty/failure yields an explicit warning, never a silent zero list. + zed: { + customResolver: async (connection) => { + try { + const result = await resolveZedModels({ + accessToken: connection.accessToken, + providerSpecificData: connection.providerSpecificData || {}, + }, { config: ZED_HOSTED_CONFIG, forceRefresh: true }); + const models = (result?.models || []) + .filter((m) => m && !m.isDisabled) + .map((m) => ({ + id: m.id, + name: m.name || m.id, + provider: m.provider, + contextLength: m.contextLength, + contextLengthInMaxMode: m.contextLengthInMaxMode, + maxOutputTokens: m.maxOutputTokens, + supportsTools: m.supportsTools, + supportsImages: m.supportsImages, + supportsThinking: m.supportsThinking, + supportsDisablingThinking: m.supportsDisablingThinking, + supportsFastMode: m.supportsFastMode, + supportsServerSideCompaction: m.supportsServerSideCompaction, + supportedEffortLevels: m.supportedEffortLevels || [], + supportsStreamingTools: m.supportsStreamingTools, + supportsParallelToolCalls: m.supportsParallelToolCalls, + })); + if (models.length > 0) return { models }; + return { models: [], warning: "Zed returned no live models." }; + } catch (error) { + console.log("Failed to fetch Zed models dynamically:", error.message); + return { models: [], warning: `Failed to fetch Zed models: ${error.message}` }; + } + }, + }, + + // Cline/ClinePass share api.cline.bot/api/v1/models. The service layer already + // handles Bearer-vs-`workos:` auth and swallows failures into null, so these follow + // the cursor direct pattern (no refreshFn) and only differ in filtering: + // cline returns the whole catalog verbatim, clinepass keeps cline-pass/* only. + cline: { + customResolver: async (connection) => { + const result = await resolveClineModels({ + accessToken: connection.accessToken, + apiKey: connection.apiKey, + }); + if (result?.models?.length) return { models: result.models }; + return { + models: getStaticProviderModels("cline"), + warning: "Cline returned no live models; falling back to static catalog.", + }; + }, + }, + clinepass: { + customResolver: async (connection) => { + const result = await resolveClinepassModels({ + accessToken: connection.accessToken, + apiKey: connection.apiKey, + }); + if (result?.models?.length) return { models: result.models }; + return { + models: getStaticProviderModels("clinepass"), + warning: "ClinePass returned no live models; falling back to static catalog.", + }; + }, + }, // Custom resolvers (non-OpenAI-shaped APIs / token-refresh flows) kiro: { diff --git a/src/app/api/providers/[id]/test/testUtils.js b/src/app/api/providers/[id]/test/testUtils.js index 3c337ffb..75d6a965 100644 --- a/src/app/api/providers/[id]/test/testUtils.js +++ b/src/app/api/providers/[id]/test/testUtils.js @@ -4,6 +4,7 @@ import { testProxyUrl } from "@/lib/network/proxyTest"; import { isOpenAICompatibleProvider, isAnthropicCompatibleProvider } from "@/shared/constants/providers"; import { getDefaultModel } from "open-sse/config/providerModels.js"; import { resolveOllamaLocalHost, PROVIDERS } from "open-sse/config/providers.js"; +import { CODEX_CLI_VERSION } from "open-sse/config/appConstants.js"; import { refreshProviderCredentials, shouldRefreshCredentials, @@ -27,7 +28,7 @@ const OAUTH_TEST_CONFIG = { method: "POST", authHeader: "Authorization", authPrefix: "Bearer ", - extraHeaders: { "Content-Type": "application/json", "originator": "codex_cli_rs", "User-Agent": "codex_cli_rs/0.136.0" }, + extraHeaders: { "Content-Type": "application/json", "originator": "codex_cli_rs", "User-Agent": `codex_cli_rs/${CODEX_CLI_VERSION}` }, // Minimal invalid body — triggers fast 400 without consuming quota body: JSON.stringify({ model: "gpt-5.3-codex", input: [], stream: false, store: false }), // 400 (bad request) means auth succeeded; only 401/403 means token is bad @@ -765,6 +766,12 @@ async function testApiKeyConnection(connection, effectiveProxy = null) { const valid = !!(data && data.user); return { valid, error: valid ? null : "Session expired — re-paste cookie" }; } + case "opencode": { + const res = await fetchWithConnectionProxy("https://opencode.ai/zen/v1/models", { + headers: { Authorization: "Bearer public", "User-Agent": "opencode/1.18.31" }, + }, effectiveProxy); + return { valid: res.ok, error: res.ok ? null : "OpenCode free tier unavailable" }; + } case "opencode-go": { const res = await fetchWithConnectionProxy("https://opencode.ai/zen/go/v1/chat/completions", { method: "POST", diff --git a/src/app/api/providers/suggested-models/filters.js b/src/app/api/providers/suggested-models/filters.js index 13454757..8babbb9b 100644 --- a/src/app/api/providers/suggested-models/filters.js +++ b/src/app/api/providers/suggested-models/filters.js @@ -26,4 +26,10 @@ export const FILTERS = { (Array.isArray(models) ? models : []) .filter((m) => m.id?.startsWith("mimo") || m.name?.toLowerCase().includes("mimo")) .map((m) => ({ id: m.id, name: m.name || m.id })), + + "airforce-free": (models) => + (Array.isArray(models) ? models : []) + .filter((m) => (m.tier === "free" || m.id?.endsWith(":free")) && m.supports_chat === true && (!m.media_type || m.media_type === "chat" || m.media_type === "text")) + .map((m) => ({ id: m.id, name: m.name || m.id, contextLength: m.context_length })) + .sort((a, b) => String(a.id).localeCompare(String(b.id))), }; diff --git a/src/app/api/v1/models/route.js b/src/app/api/v1/models/route.js index 40adc696..9ce571b4 100644 --- a/src/app/api/v1/models/route.js +++ b/src/app/api/v1/models/route.js @@ -9,9 +9,9 @@ import { getProviderConnections, getCombos, getCustomModels, getModelAliases } f import { getDisabledModels } from "@/lib/disabledModelsDb"; import { resolveKiroModels } from "open-sse/services/kiroModels.js"; import { resolveKimchiModels } from "open-sse/services/kimchiModels.js"; -import { resolveQoderModels } from "open-sse/services/qoderModels.js"; +import { resolveQoderModels, routableQoderModels } from "open-sse/services/qoderModels.js"; import { resolveCopilotModels } from "open-sse/services/copilotModels.js"; -import { resolveClinepassModels } from "open-sse/services/clinepassModels.js"; +import { resolveClinepassModels, resolveClineModels } from "open-sse/services/clinepassModels.js"; import { resolveGrokCliModels } from "open-sse/services/grokCliModels.js"; import { resolveCursorModels } from "open-sse/services/cursorModels.js"; import { resolveZedModels } from "open-sse/shared/zedAuth.js"; @@ -34,15 +34,18 @@ const LIVE_MODEL_RESOLVERS = { qoder: async (conn) => { const result = await resolveQoderModels({ accessToken: conn.accessToken, + // PAT (pt-...) connections keep the token in apiKey; without it the live + // catalog silently fails and /v1/models falls back to the static list. + apiKey: conn.apiKey, refreshToken: conn.refreshToken, email: conn.email, displayName: conn.displayName, providerSpecificData: conn.providerSpecificData || {} }); - if (!result?.models?.length) return null; - return { - models: result.models.map((m) => ({ id: m.id, name: m.name })), - }; + // Visible + hidden (enable:false) catalog keys — chat routes all of them. + const models = routableQoderModels(result); + if (!models.length) return null; + return { models: models.map((m) => ({ id: m.id, name: m.name })) }; }, kimchi: async (conn) => { const result = await resolveKimchiModels({ @@ -76,6 +79,13 @@ const LIVE_MODEL_RESOLVERS = { }); return result?.models?.length ? { models: result.models } : null; }, + cline: async (conn) => { + const result = await resolveClineModels({ + accessToken: conn.accessToken, + apiKey: conn.apiKey, + }); + return result?.models?.length ? { models: result.models } : null; + }, "grok-cli": async (conn) => { const proxy = await resolveConnectionProxyConfig(conn.providerSpecificData || {}); const result = await resolveGrokCliModels({ diff --git a/src/lib/auth/dashboardSession.js b/src/lib/auth/dashboardSession.js index b4c65aae..8f81cec5 100644 --- a/src/lib/auth/dashboardSession.js +++ b/src/lib/auth/dashboardSession.js @@ -7,6 +7,7 @@ import { DATA_DIR } from "@/lib/dataDir"; import { getSettings } from "@/lib/localDb"; const DEFAULT_PASSWORD = "123456"; +const SESSION_MAX_AGE_SEC = 24 * 60 * 60; function loadJwtSecret() { if (process.env.JWT_SECRET) return process.env.JWT_SECRET; @@ -64,6 +65,7 @@ export async function setDashboardAuthCookie(cookieStore, request, claims = {}) secure: shouldUseSecureCookie(request), sameSite: "lax", path: "/", + maxAge: SESSION_MAX_AGE_SEC, }); } diff --git a/src/lib/db/driver.js b/src/lib/db/driver.js index 050514d9..17b6dd49 100644 --- a/src/lib/db/driver.js +++ b/src/lib/db/driver.js @@ -19,6 +19,10 @@ async function tryBunSqlite() { async function tryBetterSqlite() { // Skip on Bun — better-sqlite3 native bindings unsupported if (process.versions.bun) return null; + // Skip on Node >= 24: the native addon SIGSEGVs on load there, which is a + // process-level crash the try/catch below cannot recover from. node:sqlite covers it. + const [nodeMajor] = process.versions.node.split(".").map(Number); + if (nodeMajor >= 24) return null; try { const { createBetterSqliteAdapter } = await import("./adapters/betterSqliteAdapter.js"); return createBetterSqliteAdapter(DATA_FILE); diff --git a/src/lib/db/repos/connectionsRepo.js b/src/lib/db/repos/connectionsRepo.js index 3b510965..28b419be 100644 --- a/src/lib/db/repos/connectionsRepo.js +++ b/src/lib/db/repos/connectionsRepo.js @@ -11,6 +11,28 @@ const OPTIONAL_FIELDS = [ "proxyRotationStrategy", "proxyPoolIds", ]; +const MODEL_LOCK_PREFIX = "modelLock_"; + +function resetHealthStateOnActivation(existing, patch) { + if (patch?.testStatus !== "active") return patch; + + const normalized = { + ...patch, + testStatus: "active", + lastError: Object.hasOwn(patch, "lastError") ? patch.lastError : null, + lastErrorAt: Object.hasOwn(patch, "lastErrorAt") ? patch.lastErrorAt : null, + errorCode: null, + rateLimitedUntil: null, + backoffLevel: 0, + }; + + for (const key of Object.keys(existing || {})) { + if (key.startsWith(MODEL_LOCK_PREFIX)) normalized[key] = null; + } + + return normalized; +} + function rowToConn(row) { if (!row) return null; const extra = parseJson(row.data, {}); @@ -148,7 +170,8 @@ export async function createProviderConnection(data) { // access_token: never dedup — user manages duplicates manually if (existing) { - const merged = { ...existing, ...data, updatedAt: now }; + const normalized = resetHealthStateOnActivation(existing, data); + const merged = { ...existing, ...normalized, updatedAt: now }; upsert(db, merged); result = merged; return; @@ -197,7 +220,8 @@ export async function updateProviderConnection(id, data) { const row = db.get(`SELECT * FROM providerConnections WHERE id = ?`, [id]); if (!row) { result = null; return; } const existing = rowToConn(row); - const merged = { ...existing, ...data, updatedAt: new Date().toISOString() }; + const normalized = resetHealthStateOnActivation(existing, data); + const merged = { ...existing, ...normalized, updatedAt: new Date().toISOString() }; upsert(db, merged); if (data.priority !== undefined) reorderInTx(db, existing.provider); result = merged; diff --git a/src/lib/modelCatalog/sync.js b/src/lib/modelCatalog/sync.js index 0d49c48e..973c18b3 100644 --- a/src/lib/modelCatalog/sync.js +++ b/src/lib/modelCatalog/sync.js @@ -6,7 +6,7 @@ import fs from "node:fs"; import path from "node:path"; -import { CATALOG_FILE, CATALOG_RAW_FILE, invalidateCatalog, installCatalogSource } from "open-sse/providers/catalogOverride.js"; +import { CATALOG_FILE, CATALOG_RAW_FILE, CATALOG_VERSION, invalidateCatalog, installCatalogSource } from "open-sse/providers/catalogOverride.js"; const CATALOG_URL = "https://models.dev/api.json"; const FETCH_TIMEOUT_MS = 60000; @@ -16,16 +16,14 @@ const STARTUP_DELAY_MS = 60 * 1000; // let the server boot and serve first req const RETRY_DELAY_MS = 30 * 60 * 1000; const MODALITY_BY_INPUT = { image: "vision", pdf: "pdf", audio: "audioInput", video: "videoInput" }; -// Gateways disagree about the same model, so a modality needs a majority of -// them to declare it — one reseller mislabelling a text model must not win. -const MIN_SHARE = 0.5; // Ignore limit differences below this: gateways round 200000 vs 202752. const LIMIT_TOLERANCE = 0.1; -// 9router provider id -> models.dev provider id, for context/maxOutput only. -// Providers absent here keep whatever the local pattern table resolves; names -// that already match are resolved automatically. -const PROVIDER_ALIASES = { +// 9router provider id -> models.dev provider id: the same gateway under another +// name. Both halves of the catalog are stored against the local id, so this runs +// while building rather than on every lookup. Providers absent here keep whatever +// the local pattern table resolves; names that already match need no entry. +export const PROVIDER_ALIASES = { "glm": "zai", "glm-cn": "zhipuai", "claude": "anthropic", @@ -40,7 +38,7 @@ const PROVIDER_ALIASES = { "cloudflare-ai": "cloudflare-workers-ai", }; -let state = { running: false, lastSync: null, lastError: null, lastResult: null, etag: null }; +let state = { running: false, lastSync: null, lastError: null, lastResult: null, etag: null, fileVersion: null }; let timer = null; export function getSyncState() { @@ -78,40 +76,57 @@ function slim(catalog) { return out; } -function build(catalog, entries) { - // Index once: per provider for limits, and tallied across all of them for - // modalities. - const byProvider = {}; - const tally = {}; - for (const [providerId, provider] of Object.entries(catalog)) { - const models = {}; - const counted = new Set(); - for (const [modelId, model] of Object.entries(provider?.models || {})) { - const id = baseId(modelId); - models[id] = model; - // One vote per provider: several ids can normalize to the same model - // (claude-opus-4-thinking:1024, :8192, :32768 …) and must not stack. - if (counted.has(id)) continue; - counted.add(id); - const counts = tally[id] || (tally[id] = { total: 0 }); - counts.total++; - for (const input of model?.modalities?.input || []) { - const key = MODALITY_BY_INPUT[input]; - if (key) counts[key] = (counts[key] || 0) + 1; - } - } - byProvider[providerId] = models; +export function build(catalog, entries) { + // Upstream provider id -> the local ids it belongs to, taken from the registry + // snapshot so a gateway listed upstream under another name is still filed + // under the name requests arrive with. One upstream name can back more than one + // local id (glm-cn and zhipu are both zhipuai) and each has to resolve; the + // snapshot only covers the built-in registry, so an upstream provider it does + // not mention keeps its own name. + const localIds = new Map(); + for (const { provider } of entries) { + const upstreamId = PROVIDER_ALIASES[provider] || provider; + let locals = localIds.get(upstreamId); + if (!locals) localIds.set(upstreamId, (locals = [])); + if (!locals.includes(provider)) locals.push(provider); } - // Modalities belong to the model — every gateway serving it has the same - // weights — so they are keyed by model id and shared across providers. + // Index once: the raw upstream record per provider+model for limits, and the + // modalities each gateway declares for it. + const byProvider = {}; + // Modalities are recorded per gateway upstream and gateways disagree about the + // same weights — some do not proxy images at all — so the key is provider + + // model. Keying by model id alone let short ids collide across vendors: "auto", + // "free" and "efficient" are router modes in one catalog and model names in + // another, so a router mode inherited a stranger's vision. const models = {}; - for (const [id, counts] of Object.entries(tally)) { - const declared = {}; - for (const key of Object.values(MODALITY_BY_INPUT)) { - if ((counts[key] || 0) / counts.total >= MIN_SHARE) declared[key] = true; + for (const [providerId, provider] of Object.entries(catalog)) { + const locals = localIds.get(providerId) || [providerId]; + const modelsById = {}; + const seen = new Set(); + for (const [modelId, model] of Object.entries(provider?.models || {})) { + const id = baseId(modelId); + modelsById[id] = model; + // One entry per provider+model: several upstream ids can normalize to the + // same model (claude-opus-4-thinking:1024, :8192, :32768 …) and must not + // stack their modalities. + if (seen.has(id)) continue; + seen.add(id); + const declared = {}; + for (const input of model?.modalities?.input || []) { + const key = MODALITY_BY_INPUT[input]; + if (key) declared[key] = true; + } + if (Object.keys(declared).length) { + // Filed under every local id requests arrive with, and under the upstream + // id too: a custom provider node can carry the upstream name without + // appearing in the registry snapshot, and nothing else would resolve for + // it. The reader takes whichever key it is handed. + for (const local of locals) models[`${local}:${id}`] = declared; + if (!locals.includes(providerId)) models[`${providerId}:${id}`] = declared; + } } - if (Object.keys(declared).length) models[id] = declared; + byProvider[providerId] = modelsById; } // Limits belong to the gateway — each truncates differently — so only the @@ -172,7 +187,9 @@ export async function syncModelCatalog() { state.running = true; try { const headers = { accept: "application/json" }; - if (state.etag) headers["if-none-match"] = state.etag; + // A file written by an older schema has to be rebuilt even when upstream is + // unchanged, so only ask upstream for a 304 when the file is current. + if (state.etag && state.fileVersion === CATALOG_VERSION) headers["if-none-match"] = state.etag; const response = await fetch(CATALOG_URL, { headers, signal: AbortSignal.timeout(FETCH_TIMEOUT_MS) }); let result; @@ -187,12 +204,13 @@ export async function syncModelCatalog() { const etag = response.headers.get("etag") || null; const entries = await collectEntries(); const { models, providers } = build(catalog, entries); - const serialized = JSON.stringify({ v: 1, etag, syncedAt: Date.now(), models, providers }); + const serialized = JSON.stringify({ v: CATALOG_VERSION, etag, syncedAt: Date.now(), models, providers }); writeAtomic(CATALOG_FILE, serialized); writeAtomic(CATALOG_RAW_FILE, JSON.stringify(slim(catalog))); state.etag = etag; + state.fileVersion = CATALOG_VERSION; invalidateCatalog(); result = { status: "updated", @@ -223,10 +241,13 @@ export async function syncModelCatalog() { // of re-downloading 4.3MB to be told nothing changed. function restoreEtag() { try { - state.etag = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8")).etag || null; + const parsed = JSON.parse(fs.readFileSync(CATALOG_FILE, "utf8")); + state.etag = parsed.etag || null; + state.fileVersion = parsed.v || 1; state.lastSync = fs.statSync(CATALOG_FILE).mtimeMs; } catch { state.etag = null; + state.fileVersion = null; } } diff --git a/src/lib/oauth/constants/oauth.js b/src/lib/oauth/constants/oauth.js index 8532a1dd..25be3333 100644 --- a/src/lib/oauth/constants/oauth.js +++ b/src/lib/oauth/constants/oauth.js @@ -133,6 +133,21 @@ export const FREEBUFF_CONFIG = { ...PROVIDER_OAUTH["freebuff"] }; // 3) Redirect → ${cb}?refreshToken=...&loginHost=...&isRedirect=true // 4) POST ExchangeToken {ClientID, RefreshToken, ClientSecret:"-"} → {Result.AccessToken, ExpiresAt} // 5) POST GetUserInfo (x-cloudide-token) → email/name +// Xiaomi MiMo Desktop OAuth — custom ECDH encrypted-callback flow (NOT standard OAuth2). +// 1) Client generates X25519 keypair +// 2) Browser opens ${platformUrl}/authorize?pk=&redirect_uri=http://localhost:/&kn=mimocode&key_name=... +// 3) Redirect → http://localhost:/?u= +// 4) Decrypt: ECDH(shared) → SHA256 → AES-256-GCM +// Layout: [12-byte nonce][32-byte ephemeral pubkey][ciphertext][16-byte GCM tag] +// 5) Result JSON: { uid, sk, url } +export const XIAOMI_MIMO_CONFIG = { + platformUrl: process.env.MIMO_PLATFORM_URL || "https://platform.xiaomimimo.com", + defaultBaseUrl: "https://api.xiaomimimo.com/v1", + kn: "mimocode", + callbackPath: "/", + timeoutMs: 300000, // 5 minutes +}; + export const TRAE_CONFIG = { clientId: "ono9krqynydwx5", clientSecret: "-", diff --git a/src/lib/oauth/providers/index.js b/src/lib/oauth/providers/index.js index d013d307..b4a392f3 100644 --- a/src/lib/oauth/providers/index.js +++ b/src/lib/oauth/providers/index.js @@ -114,6 +114,11 @@ export async function generateAuthData(providerName, redirectUri, meta) { flowType: provider.flowType, fixedPort: provider.fixedPort, callbackPath: provider.callbackPath || "/callback", + // Zed: surface the system_id embedded in the sign-in URL so the frontend + // can thread it through register-session → exchange → stored connection + // (exchangeTokens re-runs prepareConfig, which would otherwise mint a + // different one). Absent for every other provider — purely additive. + ...(config.systemId ? { systemId: config.systemId } : {}), }; } diff --git a/src/lib/oauth/providers/xiaomi-mimo.js b/src/lib/oauth/providers/xiaomi-mimo.js new file mode 100644 index 00000000..ff85cf7b --- /dev/null +++ b/src/lib/oauth/providers/xiaomi-mimo.js @@ -0,0 +1,123 @@ +import crypto from "crypto"; +import { XIAOMI_MIMO_CONFIG } from "../constants/oauth.js"; + +// ─────────────────────────────────────────────────────────────────────────── +// Xiaomi MiMo OAuth helpers +// Custom ECDH + AES-256-GCM encrypted-callback flow (NOT standard OAuth2). +// ─────────────────────────────────────────────────────────────────────────── + +/** + * Generate an X25519 keypair for the OAuth handshake. + * @returns {{ publicKey: string, privateKeyDer: Buffer }} + * publicKey — base64 SPKI (for the `pk` URL param) + * privateKeyDer — PKCS8 DER Buffer (for ECDH later) + */ +export function generateKeyPair() { + const { publicKey, privateKey } = crypto.generateKeyPairSync("x25519"); + + const publicKeyDer = publicKey.export({ format: "der", type: "spki" }); + // SPKI for X25519 is 44 bytes; the raw 32-byte key is the last 32 bytes. + // But the platform expects the full base64 SPKI — pass as-is. + const publicKeyB64 = publicKeyDer.toString("base64"); + + const privateKeyDer = privateKey.export({ format: "der", type: "pkcs8" }); + + return { publicKey: publicKeyB64, privateKeyDer }; +} + +/** + * Decrypt the `u` query parameter from the Xiaomi OAuth callback. + * + * Wire format (base64-decoded): + * bytes 0..11 — 12-byte AES-GCM nonce + * bytes 12..43 — 32-byte ephemeral public key (raw X25519) + * bytes 44..n-16 — ciphertext + * last 16 bytes — GCM auth tag + * + * Key derivation: SHA256(ECDH(clientPrivateKey, ephemeralPublicKey)) + * + * @param {Buffer} privateKeyDer — PKCS8 DER private key from generateKeyPair() + * @param {string} encryptedB64 — the `u` query param value (base64) + * @returns {{ uid: string, sk: string, url?: string }} + */ +export function decryptCallback(privateKeyDer, encryptedB64) { + const raw = Buffer.from(encryptedB64, "base64"); + + if (raw.length < 12 + 32 + 16 + 1) { + throw new Error(`Encrypted payload too short: ${raw.length} bytes`); + } + + const nonce = raw.subarray(0, 12); + const ephemeralPubRaw = raw.subarray(12, 44); + const ciphertextAndTag = raw.subarray(44); + const tag = ciphertextAndTag.subarray(ciphertextAndTag.length - 16); + const ciphertext = ciphertextAndTag.subarray(0, ciphertextAndTag.length - 16); + + // Reconstruct the ephemeral public key as SPKI DER for Node crypto. + // X25519 SPKI prefix: 302a300506032b656e032100 + const ephemeralPub = crypto.createPublicKey({ + key: Buffer.concat([ + Buffer.from("302a300506032b656e032100", "hex"), + ephemeralPubRaw, + ]), + format: "der", + type: "spki", + }); + + const privateKey = crypto.createPrivateKey({ + key: privateKeyDer, + format: "der", + type: "pkcs8", + }); + + const sharedSecret = crypto.diffieHellman({ privateKey, publicKey: ephemeralPub }); + const derivedKey = crypto.createHash("sha256").update(sharedSecret).digest(); + + const decipher = crypto.createDecipheriv("aes-256-gcm", derivedKey, nonce); + decipher.setAuthTag(tag); + const decrypted = Buffer.concat([decipher.update(ciphertext), decipher.final()]); + + const parsed = JSON.parse(decrypted.toString("utf-8")); + + if (!parsed || typeof parsed !== "object") { + throw new Error("Decrypted payload is not a valid object"); + } + + return { + uid: parsed.uid || null, + sk: parsed.sk || null, + url: parsed.url || XIAOMI_MIMO_CONFIG.defaultBaseUrl, + }; +} + +/** + * Build the browser authorization URL. + * @param {string} publicKey — base64 SPKI from generateKeyPair() + * @param {string} redirectUri — e.g. http://localhost:12345/ + * @param {string} [keyName] — optional stable key name + * @returns {string} + */ +export function buildAuthorizeUrl(publicKey, redirectUri, keyName) { + const params = new URLSearchParams({ + pk: publicKey, + redirect_uri: redirectUri, + kn: XIAOMI_MIMO_CONFIG.kn, + }); + if (keyName) params.set("key_name", keyName); + return `${XIAOMI_MIMO_CONFIG.platformUrl}/authorize?${params.toString()}`; +} + +/** + * Get or create a stable key name for this installation. + * Stored in the 9Router data dir so re-auth reuses the same name. + */ +export function getKeyName() { + // Use a deterministic name based on machine — avoids needing filesystem writes + // in the OAuth provider layer. The platform treats key_name as a label only. + const machineId = crypto + .createHash("sha256") + .update(`${process.platform}-${process.env.COMPUTERNAME || process.env.HOSTNAME || "unknown"}`) + .digest("hex") + .slice(0, 8); + return `9router-xmd-${machineId}`; +} diff --git a/src/lib/oauth/providers/zed.js b/src/lib/oauth/providers/zed.js index 3343976a..3f38bf87 100644 --- a/src/lib/oauth/providers/zed.js +++ b/src/lib/oauth/providers/zed.js @@ -21,11 +21,15 @@ const zed = { return { ...config, ...auth }; }, buildAuthUrl: (config, redirectUri, state) => config.authUrl, - exchangeToken: async (config, code, redirectUri, codeVerifier, state) => { + exchangeToken: async (config, code, redirectUri, codeVerifier, state, meta) => { // code = raw callback URL/query; codeVerifier = encoded private key verifier. const { userId, encryptedAccessToken } = parseZedCallbackPayload(code); const accessToken = decryptZedAccessToken(encryptedAccessToken, codeVerifier); - return { accessToken, userId, systemId: config.systemId }; + // Prefer the system_id registered for this login attempt (threaded via + // meta from register-session); fall back to the prepared config. Never + // mint a fresh one here — exchangeTokens re-runs prepareConfig, which + // would otherwise store a system_id unrelated to the zed.dev login. + return { accessToken, userId, systemId: meta?.systemId || config.systemId }; }, postExchange: async (tokens) => { const credentials = { diff --git a/src/lib/oauth/utils/server.js b/src/lib/oauth/utils/server.js index 56eb67b1..83de1821 100644 --- a/src/lib/oauth/utils/server.js +++ b/src/lib/oauth/utils/server.js @@ -648,9 +648,15 @@ let zedProxyTimeout = null; let zedProxyPort = null; let zedSession = null; -export function registerZedSession({ state, codeVerifier }) { +export function registerZedSession({ state, codeVerifier, systemId }) { if (!state || !codeVerifier) return false; - zedSession = { state, codeVerifier, status: "pending", createdAt: Date.now() }; + zedSession = { + state, + codeVerifier, + systemId: systemId || null, + status: "pending", + createdAt: Date.now(), + }; return true; } export function getZedSessionStatus(state) { @@ -665,6 +671,10 @@ export function clearZedSession(state) { export function startZedProxy(preferredPort = 0) { return new Promise((resolve) => { if (zedProxyServer) { + // Reuse the live listener, but renew its idle timeout so a previous + // flow's deadline can never kill the flow that just adopted the port. + if (zedProxyTimeout) clearTimeout(zedProxyTimeout); + zedProxyTimeout = setTimeout(() => { console.log("[Zed proxy] timeout, stopping"); stopZedProxy(); }, ZED_HOSTED_CONFIG.oauthTimeoutMs); resolve({ success: true, port: zedProxyPort, callbackUrl: `http://127.0.0.1:${zedProxyPort}/` }); return; } @@ -694,13 +704,34 @@ export function startZedProxy(preferredPort = 0) { res.end(renderCodexResultPage(false, "Cross-origin callback rejected")); return; } + // A genuine Zed redirect always carries user_id + access_token. Anything + // else (probe, prefetch, stray navigation, favicon-style miss) is NOT + // the callback: answer without touching the session and WITHOUT + // stopping the server, so the real redirect can still land afterwards. + const qp = url.searchParams; + const hasZedParams = + qp.has("user_id") || qp.has("userId") || + qp.has("access_token") || qp.has("accessToken") || qp.has("token"); + if (!hasZedParams) { + console.log(`[Zed proxy] ignoring non-callback ${req.method} ${url.pathname} (session kept, server kept)`); + res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderCodexResultPage(false, "Waiting for Zed sign-in — this request carried no login data.")); + return; + } // Pass raw callback path+query to exchangeTokens → parseZedCallbackPayload. // codeVerifier carries the encoded RSA private key for decryption. const rawCallback = url.search ? `${url.pathname}?${url.searchParams.toString()}` : url.pathname; try { const { exchangeTokens } = await import("../providers.js"); const { createProviderConnection } = await import("@/models"); - const tokenData = await exchangeTokens("zed", rawCallback, null, session.codeVerifier, session.state); + const tokenData = await exchangeTokens( + "zed", + rawCallback, + null, + session.codeVerifier, + session.state, + session.systemId ? { systemId: session.systemId } : undefined, + ); const connection = await createProviderConnection({ provider: "zed", authType: "oauth", @@ -712,13 +743,16 @@ export function startZedProxy(preferredPort = 0) { session.email = connection.email; res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); res.end(renderCodexResultPage(true, "You can close this window.")); + stopZedProxy(); } catch (err) { session.status = "error"; session.error = err.message; res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); res.end(renderCodexResultPage(false, err.message)); - } finally { - stopZedProxy(); + // Intentionally NOT stopping here: the failure may belong to a + // superseded attempt (e.g. an older popup landing after "Try Again" + // registered a new keypair). The live attempt's genuine callback must + // still land. The idle timeout + modal close bound the listener. } }); const tryPort = Number(preferredPort) || 0; @@ -755,3 +789,185 @@ export function stopZedProxy() { zedProxyPort = null; } +// ─────────────────────────────────────────────────────────────────────────── +// Xiaomi MiMo Desktop OAuth callback proxy +// Receives the ECDH-encrypted `u` param, decrypts it, stores the session. +// ─────────────────────────────────────────────────────────────────────────── + +let xiaomiMimoProxyServer = null; +let xiaomiMimoProxyPort = null; +let xiaomiMimoProxyTimeout = null; + +const xiaomiMimoSessions = new Map(); + +export function registerXiaomiMimoSession({ state, privateKeyDer }) { + if (!state || !privateKeyDer) return false; + xiaomiMimoSessions.set(state, { + privateKeyDer, + status: "pending", + createdAt: Date.now(), + }); + return true; +} + +export function getXiaomiMimoSessionStatus(state) { + const s = xiaomiMimoSessions.get(state); + if (!s) return null; + // Don't leak the private key to the client + return { status: s.status, result: s.result || null, error: s.error || null }; +} + +export function clearXiaomiMimoSession(state) { + xiaomiMimoSessions.delete(state); +} + +function renderXiaomiMimoResultPage(success, message) { + const color = success ? "#22c55e" : "#ef4444"; + const icon = success ? "✓" : "✗"; + const title = success ? "Authentication Successful" : "Authentication Failed"; + return ` + +${title} + + + +

+
${icon}
+

${title}

+

${message || (success ? "You can close this tab and return to 9Router." : "Please try again.")}

+ ${success ? "" : ""} +
+ +`; +} + +/** + * Start the Xiaomi Desktop OAuth callback proxy. + * @returns {Promise<{success: boolean, port?: number, callbackUrl?: string, reason?: string}>} + */ +export function startXiaomiMimoProxy() { + return new Promise((resolve) => { + if (xiaomiMimoProxyServer) { + resolve({ + success: true, + port: xiaomiMimoProxyPort, + callbackUrl: `http://127.0.0.1:${xiaomiMimoProxyPort}/`, + }); + return; + } + + const server = http.createServer(async (req, res) => { + // Origin guard + if (!isLoopbackOrigin(req.headers.origin)) { + res.writeHead(403); + res.end("Forbidden"); + return; + } + + const url = new URL(req.url, "http://127.0.0.1"); + const u = url.searchParams.get("u"); + + if (!u) { + res.writeHead(400, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(false, "Missing encrypted payload (u parameter).")); + return; + } + + // Try each pending session's private key — the callback URL carries no + // state param, so we attempt decryption with every pending key. + const pendingSessions = [...xiaomiMimoSessions.entries()] + .filter(([, s]) => s.status === "pending"); + + if (pendingSessions.length === 0) { + res.writeHead(500, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(false, "No active OAuth session. Please restart the login flow.")); + return; + } + + try { + const { decryptCallback } = await import("../providers/xiaomi-mimo.js"); + let result = null; + let matchedState = null; + + for (const [state, session] of pendingSessions) { + try { + result = decryptCallback(session.privateKeyDer, u); + matchedState = state; + break; + } catch { + // Wrong key for this session — try next + } + } + + if (!result || !matchedState) { + throw new Error("Could not decrypt with any pending session key"); + } + + if (!result.sk) { + throw new Error("Decrypted payload missing sk (API key)"); + } + + // Store result only in the matched session + const session = xiaomiMimoSessions.get(matchedState); + if (session) { + session.status = "done"; + session.result = { + uid: result.uid, + accessToken: result.sk, + baseUrl: result.url || "https://api.xiaomimimo.com/v1", + }; + } + + res.writeHead(200, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(true, "Xiaomi account linked. You can close this tab.")); + console.log("[xiaomi-mimo oauth] callback decrypted, uid:", result.uid); + } catch (err) { + console.error("[xiaomi-mimo oauth] decrypt failed:", err.message); + for (const [, session] of pendingSessions) { + session.status = "error"; + session.error = err.message; + } + res.writeHead(400, { "Content-Type": "text/html; charset=utf-8" }); + res.end(renderXiaomiMimoResultPage(false, `Decryption failed: ${err.message}`)); + } + }); + + server.on("error", (err) => { + console.log("[xiaomi-mimo oauth] listen error:", err.message); + resolve({ success: false, reason: err.message }); + }); + + server.listen(0, "127.0.0.1", () => { + xiaomiMimoProxyServer = server; + xiaomiMimoProxyPort = server.address().port; + xiaomiMimoProxyTimeout = setTimeout(() => { + console.log("[xiaomi-mimo oauth] timeout, stopping"); + stopXiaomiMimoProxy(); + }, 300000); + console.log(`[xiaomi-mimo oauth] listening on port ${xiaomiMimoProxyPort}`); + resolve({ + success: true, + port: xiaomiMimoProxyPort, + callbackUrl: `http://127.0.0.1:${xiaomiMimoProxyPort}/`, + }); + }); + }); +} + +export function stopXiaomiMimoProxy() { + console.log(`[xiaomi-mimo oauth] stopping (port ${xiaomiMimoProxyPort || "-"})`); + if (xiaomiMimoProxyTimeout) { clearTimeout(xiaomiMimoProxyTimeout); xiaomiMimoProxyTimeout = null; } + if (xiaomiMimoProxyServer) { xiaomiMimoProxyServer.close(); xiaomiMimoProxyServer = null; } + xiaomiMimoProxyPort = null; + // No callback can arrive once the listener is down, so drop every pending + // session — each holds an X25519 private key and they would otherwise + // accumulate for the process lifetime (one per /authorize call). + xiaomiMimoSessions.clear(); +} + diff --git a/src/shared/components/ModelSelectModal.js b/src/shared/components/ModelSelectModal.js index 2cbc11a2..20e17d95 100644 --- a/src/shared/components/ModelSelectModal.js +++ b/src/shared/components/ModelSelectModal.js @@ -20,6 +20,54 @@ const PROVIDER_ORDER = [ // Providers that need no auth — always show in model selector const NO_AUTH_PROVIDER_IDS = Object.keys(FREE_PROVIDERS).filter(id => FREE_PROVIDERS[id].noAuth); +// Providers with per-account live catalogs via /api/providers/[id]/models. +// Static registry stays as fallback when live fetch fails or is empty. +const LIVE_CATALOG_PROVIDERS = ["cursor", "cline", "clinepass"]; + +// Fetch a provider's account-scoped catalog for every active connection and merge +// the results. Entries collapse by model id on purpose: two connections of the +// same provider produce the same picker value (`alias/id`), so keeping the first +// avoids duplicate rows. There is no per-connection metadata to preserve beyond +// {id,name}. Empty array means "nothing live" so callers keep the static fallback. +function useLiveProviderModels(isOpen, connectionIds, label) { + const [models, setModels] = useState([]); + const idsKey = (connectionIds ?? []).join("|"); + + useEffect(() => { + const ids = idsKey ? idsKey.split("|") : []; + if (!isOpen || ids.length === 0) { + setModels([]); + return undefined; + } + + let cancelled = false; + Promise.all(ids.map(async (connectionId) => { + const response = await fetch(`/api/providers/${connectionId}/models`, { cache: "no-store" }); + if (!response.ok) return []; + const data = await response.json(); + return Array.isArray(data.models) ? data.models : []; + })) + .then((modelLists) => { + if (cancelled) return; + const seen = new Set(); + setModels(modelLists.flat().filter((model) => { + if (!model?.id || seen.has(model.id)) return false; + seen.add(model.id); + return true; + })); + }) + .catch((error) => { + // Do not hide the static fallback when the account catalog is unavailable. + console.warn(`Unable to load ${label} models for selector:`, error); + if (!cancelled) setModels([]); + }); + + return () => { cancelled = true; }; + }, [isOpen, idsKey, label]); + + return models; +} + export default function ModelSelectModal({ isOpen, onClose, @@ -49,48 +97,25 @@ export default function ModelSelectModal({ const [providerNodes, setProviderNodes] = useState([]); const [customModels, setCustomModels] = useState([]); const [disabledModels, setDisabledModels] = useState({}); - const [cursorModels, setCursorModels] = useState([]); - - // Cursor exposes the usable catalog per account. Keep the static catalog only - // as a fallback, since it quickly becomes stale and different accounts can - // have different model entitlements. - const cursorConnectionIds = useMemo( - () => activeProviders - .filter((provider) => provider.provider === "cursor" && provider.id) - .map((provider) => provider.id), - [activeProviders], - ); - - useEffect(() => { - if (!isOpen || cursorConnectionIds.length === 0) { - setCursorModels([]); - return undefined; + // Cursor and Cline expose the usable catalog per account, so the static catalog is + // kept only as a fallback: it goes stale quickly and entitlements differ per account. + // Single map driven by LIVE_CATALOG_PROVIDERS so the constant cannot drift + // from the memos below; per-provider arrays stay referentially stable unless + // activeProviders itself changes. + const liveConnectionIdsByProvider = useMemo(() => { + const map = Object.fromEntries(LIVE_CATALOG_PROVIDERS.map((id) => [id, []])); + for (const p of activeProviders) { + if (p?.id && Object.prototype.hasOwnProperty.call(map, p.provider)) map[p.provider].push(p.id); } + return map; + }, [activeProviders]); + const cursorConnectionIds = liveConnectionIdsByProvider.cursor; + const clineConnectionIds = liveConnectionIdsByProvider.cline; + const clinepassConnectionIds = liveConnectionIdsByProvider.clinepass; - let cancelled = false; - Promise.all(cursorConnectionIds.map(async (connectionId) => { - const response = await fetch(`/api/providers/${connectionId}/models`, { cache: "no-store" }); - if (!response.ok) return []; - const data = await response.json(); - return Array.isArray(data.models) ? data.models : []; - })) - .then((modelLists) => { - if (cancelled) return; - const seen = new Set(); - setCursorModels(modelLists.flat().filter((model) => { - if (!model?.id || seen.has(model.id)) return false; - seen.add(model.id); - return true; - })); - }) - .catch((error) => { - // Do not hide the static fallback when the account catalog is unavailable. - console.warn("Unable to load Cursor models for selector:", error); - if (!cancelled) setCursorModels([]); - }); - - return () => { cancelled = true; }; - }, [isOpen, cursorConnectionIds]); + const cursorModels = useLiveProviderModels(isOpen, cursorConnectionIds, "Cursor"); + const clineModels = useLiveProviderModels(isOpen, clineConnectionIds, "Cline"); + const clinepassModels = useLiveProviderModels(isOpen, clinepassConnectionIds, "ClinePass"); const fetchCombos = async () => { try { @@ -323,8 +348,9 @@ export default function ModelSelectModal({ hasModels: mergedModels.length > 0, }; } else { - const hardcodedModels = providerId === "cursor" && cursorModels.length > 0 - ? cursorModels + const liveModels = providerId === "cursor" ? cursorModels : providerId === "cline" ? clineModels : providerId === "clinepass" ? clinepassModels : []; + const hardcodedModels = liveModels.length > 0 + ? liveModels : getModelsByProviderId(providerId); const hardcodedIds = new Set(hardcodedModels.map((m) => m.id)); @@ -394,7 +420,7 @@ export default function ModelSelectModal({ }); return groups; - }, [filteredActiveProviders, modelAliases, allProviders, providerNodes, customModels, disabledModels, kindFilter, activeProviders, cursorModels]); + }, [filteredActiveProviders, modelAliases, allProviders, providerNodes, customModels, disabledModels, kindFilter, activeProviders, cursorModels, clineModels, clinepassModels]); // Filter combos by search query (and hide combos when kindFilter is set — combos are LLM-only by design) const filteredCombos = useMemo(() => { diff --git a/src/shared/components/OAuthModal.js b/src/shared/components/OAuthModal.js index 85ca8622..84964e60 100644 --- a/src/shared/components/OAuthModal.js +++ b/src/shared/components/OAuthModal.js @@ -50,6 +50,19 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, const popupRef = useRef(null); const pollingAbortRef = useRef(false); const openedRef = useRef(false); + // Proxy-flow session ledger: which provider's proxy THIS modal session + // started, and whether its stop was already sent. Every stop-proxy call is + // gated on this — parent re-renders can never spam it, and a close stops + // the owned proxy exactly once. + const flowRef = useRef({ proxyStarted: false, proxyProvider: null, stopSent: false }); + // Parent callbacks are stored in refs so effect/callback identities stay + // stable across parent re-renders (the page passes fresh inline closures). + // Synced by the ref-sync effect below (placed after all callbacks are + // defined); the open effect then depends only on stable primitives. + const onSuccessRef = useRef(onSuccess); + const onCloseRef = useRef(onClose); + const isOpenRef = useRef(isOpen); + const startOAuthFlowRef = useRef(null); const { copied, copy } = useCopyToClipboard(); // State for client-only values to avoid hydration mismatch @@ -81,6 +94,9 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, redirectUri: authData.redirectUri, codeVerifier: authData.codeVerifier, state, + // Zed: thread the login attempt's system_id so the stored + // connection keeps the id sent to zed.dev (see register-session). + ...(authData.systemId ? { systemId: authData.systemId } : {}), ...(oauthMeta ? { meta: oauthMeta } : {}), }), }); @@ -89,12 +105,12 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); } catch (err) { setError(err.message); setStep("error"); } - }, [authData, provider, onSuccess, oauthMeta]); + }, [authData, provider, oauthMeta]); const completeXaiManualCode = useCallback(async (code) => { if (!authData?.state) return; @@ -108,12 +124,12 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); } catch (err) { setError(err.message); setStep("error"); } - }, [authData, onSuccess]); + }, [authData]); // Poll for device code token const startPolling = useCallback(async (deviceCode, codeVerifier, interval, extraData, deadlineMs) => { @@ -155,7 +171,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, pollingAbortRef.current = true; // Stop polling immediately setStep("success"); setPolling(false); - onSuccess?.(); + onSuccessRef.current?.(); return; } @@ -177,9 +193,19 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, setError("Authorization timeout"); setStep("error"); setPolling(false); - }, [provider, onSuccess]); + }, [provider]); - // Trae/Windsurf proxy OAuth flow: dynamic-port local callback → auto exchange. + // Stop the proxy owned by THIS modal session, at most once. Re-renders, + // repeated closes, and post-completion calls are all no-ops by construction. + const stopOwnedProxy = useCallback(() => { + const flow = flowRef.current; + if (flow.proxyStarted && !flow.stopSent && flow.proxyProvider) { + flow.stopSent = true; + fetch(`/api/oauth/${flow.proxyProvider}/stop-proxy`).catch(() => {}); + } + }, []); + + // Trae/Windsurf/Zed proxy OAuth flow: dynamic-port local callback → auto exchange. const startProxyFlow = useCallback(async (providerId) => { // 1. Start the local callback server (returns a dynamic port + callback URL). const startRes = await fetch(`/api/oauth/${providerId}/start-proxy`); @@ -187,31 +213,61 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (!startRes.ok || !startData.success || !startData.callbackUrl) { throw new Error(startData.reason || startData.error || `Failed to start ${providerId} callback server`); } + // Take ownership immediately so a close during the remaining flight still + // cleans this proxy up (via the close effect or the abort below). + flowRef.current.proxyStarted = true; + flowRef.current.proxyProvider = providerId; + flowRef.current.stopSent = false; + if (!isOpenRef.current) { + stopOwnedProxy(); + return; + } // 2. Build the authorize URL with redirect_uri = proxy callback URL. const authorizeUrl = new URL(`/api/oauth/${providerId}/authorize`, window.location.origin); authorizeUrl.searchParams.set("redirect_uri", startData.callbackUrl); const authRes = await fetch(authorizeUrl); const authData = await authRes.json(); - if (!authRes.ok) throw new Error(authData.error); + if (!authRes.ok) { + stopOwnedProxy(); + throw new Error(authData.error); + } + if (!isOpenRef.current) { + stopOwnedProxy(); + return; + } // 3. Register the session so the proxy can match the incoming callback. - // Zed also passes code_verifier (encodes the RSA private key for decrypt); - // sent via POST body so the private key never lands in URL/query logs. + // Zed also passes code_verifier (encodes the RSA private key for decrypt) + // + systemId; sent via POST body so secrets never land in URL/query logs. const regBody = { state: authData.state }; if (authData.codeVerifier) regBody.codeVerifier = authData.codeVerifier; - await fetch(`/api/oauth/${providerId}/register-session`, { + if (authData.systemId) regBody.systemId = authData.systemId; + const regRes = await fetch(`/api/oauth/${providerId}/register-session`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify(regBody), }); + let regData = null; + try { + regData = await regRes.json(); + } catch { + regData = null; + } + if (!regRes.ok || regData?.success === false) { + stopOwnedProxy(); + throw new Error(regData?.error || "Failed to register login session; please retry"); + } + if (!isOpenRef.current) return; // closed mid-flight: close effect owns cleanup now // 4. Open popup; proxy auto-exchanges on callback, modal polls poll-status. setAuthData({ ...authData, proxyProvider: providerId }); setStep("waiting"); popupRef.current = window.open(authData.authUrl, "oauth_popup", "width=600,height=700"); if (!popupRef.current) setStep("input"); // popup blocked → fall back to manual paste - }, []); + }, [stopOwnedProxy]); - // Start OAuth flow - const startOAuthFlow = useCallback(async () => { + // Start OAuth flow (plain function by design: it is only invoked from the + // open effect via ref and from user actions, so memoization would only add + // an identity that re-triggers effects on every parent re-render). + const startOAuthFlow = async () => { if (!provider) return; try { setError(null); @@ -357,6 +413,14 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, setAuthData({ ...data, redirectUri, codexServerSide, xaiServerSide }); + // Take ownership of server-side proxies so close stops them exactly once + // (replaces the per-provider stop branches; same behavior, one ledger). + if ((provider === "codex" && codexProxyActive) || (provider === "xai" && xaiProxyActive)) { + flowRef.current.proxyStarted = true; + flowRef.current.proxyProvider = provider; + flowRef.current.stopSent = false; + } + // Guard: device_code providers return authUrl:null from /authorize. Never window.open(null) // (browsers coerce it to the relative path ".../null"). if (!data.authUrl) { @@ -397,49 +461,55 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, setError(err.message); setStep("error"); } - }, [provider, isLocalhost, startPolling, oauthMeta, idcConfig, authMode, startProxyFlow]); + }; - // Reset state and start OAuth when modal opens + // Sync latest props/flow into refs after every render (no dep array). + // The open effect below then depends only on stable primitives. useEffect(() => { - if (isOpen && provider) { - // Guard against StrictMode/effect re-runs auto-opening multiple tabs. - if (openedRef.current) return; - openedRef.current = true; - setAuthData(null); - setCallbackUrl(""); - setError(null); - setIsDeviceCode(false); - setDeviceData(null); - setPolling(false); - setAuthMode("browser"); - setPasteToken(""); - setIdeStatus(null); - pollingAbortRef.current = false; - // Best-effort IDE detection for paste-token providers (Trae/Windsurf) - if (PASTE_TOKEN_PROVIDERS[provider]) { - fetch(`/api/oauth/${provider}/ide-status`) - .then((r) => r.json()) - .then((data) => setIdeStatus(data)) - .catch(() => setIdeStatus({ installed: false, path: null })); - } - startOAuthFlow(); - } else if (!isOpen) { - // Abort polling and cleanup proxy when modal closes - pollingAbortRef.current = true; - openedRef.current = false; - if (provider === "codex") { - fetch("/api/oauth/codex/stop-proxy").catch(() => {}); - } else if (provider === "xai") { - fetch("/api/oauth/xai/stop-proxy").catch(() => {}); - } else if (provider === "trae") { - fetch("/api/oauth/trae/stop-proxy").catch(() => {}); - } else if (provider === "windsurf") { - fetch("/api/oauth/windsurf/stop-proxy").catch(() => {}); - } else if (provider === "zed") { - fetch("/api/oauth/zed/stop-proxy").catch(() => {}); - } + onSuccessRef.current = onSuccess; + onCloseRef.current = onClose; + isOpenRef.current = isOpen; + startOAuthFlowRef.current = startOAuthFlow; + }); + + // Reset state and start OAuth when modal opens — exactly once per open. + // Guarded by openedRef so StrictMode/effect re-runs never open extra tabs. + useEffect(() => { + if (!isOpen || !provider) return; + if (openedRef.current) return; + openedRef.current = true; + setAuthData(null); + setCallbackUrl(""); + setError(null); + setIsDeviceCode(false); + setDeviceData(null); + setPolling(false); + setAuthMode("browser"); + setPasteToken(""); + setIdeStatus(null); + pollingAbortRef.current = false; + flowRef.current = { proxyStarted: false, proxyProvider: null, stopSent: false }; + // Best-effort IDE detection for paste-token providers (Trae/Windsurf) + if (PASTE_TOKEN_PROVIDERS[provider]) { + fetch(`/api/oauth/${provider}/ide-status`) + .then((r) => r.json()) + .then((data) => setIdeStatus(data)) + .catch(() => setIdeStatus({ installed: false, path: null })); } - }, [isOpen, provider, startOAuthFlow]); + startOAuthFlowRef.current(); + }, [isOpen, provider]); + + // Cleanup when the modal closes: abort polling and stop the proxy THIS + // session started, exactly once. Deps are stable primitives, so unrelated + // parent re-renders cannot reach the stop call (previously every parent + // render re-fired stop-proxy while the modal was closed). + useEffect(() => { + if (isOpen) return; + pollingAbortRef.current = true; + openedRef.current = false; + stopOwnedProxy(); + flowRef.current = { proxyStarted: false, proxyProvider: null, stopSent: false }; + }, [isOpen, provider, stopOwnedProxy]); // Server-side proxy mode (codex/xai fixed-port + trae/windsurf dynamic-port): // poll status until the proxy auto-exchanges and saves the connection. @@ -468,7 +538,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, if (data.status === "done") { callbackProcessedRef.current = true; setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); return; } if (data.status === "error") { @@ -490,7 +560,7 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, }; setTimeout(tick, POLL_INTERVAL_MS); return () => { cancelled = true; }; - }, [authData, onSuccess]); + }, [authData]); // Listen for OAuth callback via multiple methods useEffect(() => { @@ -590,23 +660,31 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, const data = await res.json(); if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); return; } const input = callbackUrl.trim(); - // Trae/Windsurf proxy flow fallback (popup blocked): paste the full callback URL + // Trae/Windsurf/Zed proxy flow fallback (popup blocked): paste the full callback URL if (PROXY_OAUTH_PROVIDERS.has(provider) && input) { const res = await fetch(`/api/oauth/${provider}/exchange`, { method: "POST", headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ code: input, state: authData?.state }), + body: JSON.stringify({ + code: input, + state: authData?.state, + // Zed manual fallback needs the same attempt material as the + // automatic path (redirectUri + RSA verifier + system_id). + ...(authData?.redirectUri ? { redirectUri: authData.redirectUri } : {}), + ...(authData?.codeVerifier ? { codeVerifier: authData.codeVerifier } : {}), + ...(authData?.systemId ? { systemId: authData.systemId } : {}), + }), }); const data = await res.json(); if (!res.ok) throw new Error(data.error); setStep("success"); - onSuccess?.(); + onSuccessRef.current?.(); return; } @@ -653,21 +731,13 @@ export default function OAuthModal({ isOpen, provider, providerInfo, onSuccess, } }; - // Clear session on modal close + cleanup proxy + // Clear session on modal close + cleanup proxy (idempotent: the owned + // proxy is stopped at most once across effect-close, button-close, and + // Escape/backdrop-close — all funnel through here or the close effect). const handleClose = useCallback(() => { - if (provider === "codex") { - fetch("/api/oauth/codex/stop-proxy").catch(() => {}); - } else if (provider === "xai") { - fetch("/api/oauth/xai/stop-proxy").catch(() => {}); - } else if (provider === "trae") { - fetch("/api/oauth/trae/stop-proxy").catch(() => {}); - } else if (provider === "windsurf") { - fetch("/api/oauth/windsurf/stop-proxy").catch(() => {}); - } else if (provider === "zed") { - fetch("/api/oauth/zed/stop-proxy").catch(() => {}); - } - onClose(); - }, [onClose, provider]); + stopOwnedProxy(); + onCloseRef.current(); + }, [stopOwnedProxy]); if (!provider || !providerInfo) return null; const isXaiProvider = provider === "xai"; diff --git a/src/shared/components/Sidebar.js b/src/shared/components/Sidebar.js index fa52dc2c..37c58b8b 100644 --- a/src/shared/components/Sidebar.js +++ b/src/shared/components/Sidebar.js @@ -304,6 +304,9 @@ export default function Sidebar({ onClose }) { computer 9Remote + {/* + New + */} {/* 9English */} diff --git a/src/shared/components/XiaomiMimoAuthModal.js b/src/shared/components/XiaomiMimoAuthModal.js new file mode 100644 index 00000000..e501698f --- /dev/null +++ b/src/shared/components/XiaomiMimoAuthModal.js @@ -0,0 +1,276 @@ +"use client"; + +import { useState, useEffect } from "react"; +import PropTypes from "prop-types"; +import { Modal, Button } from "@/shared/components"; + +/** + * Xiaomi MiMo Auth Modal + * + * Auto-imports credentials from the local Xiaomi MiMo Desktop auth.json (~/.local/share/mimocode/auth.json). + * If auto-import fails, offers a one-click browser OAuth fallback. + * Reached only via the "Connect with OAuth" button — the API-key path uses the + * standard Add API Key modal, since Xiaomi MiMo supports both auth modes. + */ +export default function XiaomiMimoAuthModal({ isOpen, onSuccess, onClose }) { + const [phase, setPhase] = useState("detecting"); // detecting | found | not-found | importing | error + const [detectResult, setDetectResult] = useState(null); + const [error, setError] = useState(null); + const [oauthUrl, setOauthUrl] = useState(null); + const [oauthState, setOauthState] = useState(null); + + // Auto-detect local credentials when modal opens + useEffect(() => { + if (!isOpen) return; + let cancelled = false; + + (async () => { + setPhase("detecting"); + setError(null); + setDetectResult(null); + setOauthUrl(null); + + try { + const res = await fetch("/api/oauth/xiaomi-mimo/auto-import"); + const data = await res.json(); + if (cancelled) return; + + if (data.found && data.apiKey) { + setDetectResult(data); + setPhase("found"); + } else { + setPhase("not-found"); + setError(data.error || "Xiaomi MiMo Desktop credentials not found on this machine."); + } + } catch { + if (!cancelled) { + setPhase("not-found"); + setError("Failed to read local Xiaomi MiMo Desktop credentials."); + } + } + })(); + + return () => { cancelled = true; }; + }, [isOpen]); + + // Import the auto-detected key + const handleImport = async () => { + if (!detectResult?.apiKey) return; + setPhase("importing"); + setError(null); + + try { + const res = await fetch("/api/oauth/xiaomi-mimo/api-key", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + apiKey: detectResult.apiKey, + uid: detectResult.uid, + baseUrl: detectResult.baseUrl, + mimoPassToken: detectResult.mimoPassToken || null, + mimoUserId: detectResult.mimoUserId || null, + mimoCUserId: detectResult.mimoCUserId || null, + }), + }); + const data = await res.json(); + + if (!res.ok || !data.success) { + throw new Error(data.error || "Import failed"); + } + + onSuccess?.(data.connection); + onClose(); + } catch (err) { + setPhase("found"); + setError(err.message); + } + }; + + // Start browser OAuth fallback + const handleStartOAuth = async () => { + setError(null); + try { + const state = crypto.randomUUID(); + const res = await fetch(`/api/oauth/xiaomi-mimo/authorize?state=${state}`); + const data = await res.json(); + if (data.authorizeUrl) { + setOauthUrl(data.authorizeUrl); + setOauthState(data.state); + window.open(data.authorizeUrl, "_blank", "width=600,height=700"); + } else { + throw new Error(data.error || "Failed to start OAuth"); + } + } catch (err) { + setError(err.message); + } + }; + + // Poll OAuth result + const handlePollOAuth = async () => { + if (!oauthState) return; + setError(null); + try { + const res = await fetch(`/api/oauth/xiaomi-mimo/poll-status?state=${oauthState}`); + const data = await res.json(); + + if (data.status === "done" && data.result) { + // Exchange to create the connection + const exRes = await fetch("/api/oauth/xiaomi-mimo/exchange", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ state: oauthState }), + }); + const exData = await exRes.json(); + if (exData.success) { + onSuccess?.(exData.connection); + onClose(); + } else { + throw new Error(exData.error || "Exchange failed"); + } + } else if (data.status === "error") { + throw new Error(data.error || "OAuth failed"); + } else { + setError("Authorization not completed yet. Finish in the browser, then click Check Again."); + } + } catch (err) { + setError(err.message); + } + }; + + return ( + +
+ {/* Detecting */} + {phase === "detecting" && ( +
+
+ + progress_activity + +
+

Reading local credentials...

+

+ Checking ~/.local/share/mimocode/auth.json +

+
+ )} + + {/* Found — one-click import */} + {phase === "found" && detectResult && ( + <> +
+
+ check_circle +
+

Xiaomi MiMo Desktop credentials found!

+

+ UID: {detectResult.uid || "—"} · Source: {detectResult.source?.split(/[\\/]/).pop()} +

+
+
+
+ + {error && ( +
+

{error}

+
+ )} + +
+ + +
+ + )} + + {/* Importing */} + {phase === "importing" && ( +
+
+ + progress_activity + +
+

Connecting...

+
+ )} + + {/* Not found — offer OAuth fallback */} + {phase === "not-found" && ( + <> +
+
+ info +
+

Local credentials not found

+

{error}

+

+ Make sure Xiaomi MiMo Desktop is installed and you are signed in, then retry. + Or sign in via browser below. +

+
+
+
+ + {!oauthUrl ? ( +
+ + +
+ ) : ( +
+
+

+ Browser opened. Complete the Xiaomi sign-in, then click{" "} + Check Again. +

+
+
+ + +
+
+ )} + + )} +
+
+ ); +} + +XiaomiMimoAuthModal.propTypes = { + isOpen: PropTypes.bool.isRequired, + onSuccess: PropTypes.func, + onClose: PropTypes.func.isRequired, +}; diff --git a/src/shared/components/index.js b/src/shared/components/index.js index c04453a9..3d508e91 100644 --- a/src/shared/components/index.js +++ b/src/shared/components/index.js @@ -28,6 +28,7 @@ export { default as KiroAuthModal } from "./KiroAuthModal"; export { default as KiroOAuthWrapper } from "./KiroOAuthWrapper"; export { default as KiroSocialOAuthModal } from "./KiroSocialOAuthModal"; export { default as CursorAuthModal } from "./CursorAuthModal"; +export { default as XiaomiMimoAuthModal } from "./XiaomiMimoAuthModal"; export { default as IFlowCookieModal } from "./IFlowCookieModal"; export { default as GitLabAuthModal } from "./GitLabAuthModal"; export { default as EditConnectionModal } from "./EditConnectionModal"; diff --git a/src/shared/constants/cliTools.js b/src/shared/constants/cliTools.js index 9a388866..a6bb4685 100644 --- a/src/shared/constants/cliTools.js +++ b/src/shared/constants/cliTools.js @@ -30,32 +30,6 @@ export const MITM_TOOLS = { { id: "gemini-3-flash", name: "Gemini 3 Flash (Command)", alias: "gemini-3-flash" }, ], }, - copilot: { - id: "copilot", - name: "GitHub Copilot", - image: "/providers/copilot.png", - color: "#1F6FEB", - description: "GitHub Copilot IDE with MITM", - configType: "mitm", - mitmDomain: "api.individual.githubcopilot.com", - modelAliases: ["gpt-5-mini", "gpt-5.4-nano", "claude-haiku-4.5", "gpt-4o", "gpt-4.1"], - defaultModels: [ - // Verified via live MITM passthrough capture of the GitHub Copilot CLI: its model - // picker offers "GPT-5 mini" (default → wire id "gpt-5-mini"), "Claude Haiku 4.5" - // ("claude-haiku-4.5") and "Auto". "Auto" is NOT a wire id — Copilot dispatches - // concrete models dynamically (observed "gpt-5.4-nano" for light tasks and - // "claude-haiku-4.5"), so it needs no slot of its own. Without a slot for - // gpt-5-mini / gpt-5.4-nano, getMappedModel returns null and the /chat/completions - // call is passed through to GitHub Copilot instead of the configured provider — - // and gpt-5-mini is the CLI default, so the primary turn leaks (same class as the - // Kiro "auto" misrouting). gpt-4o / gpt-4.1 are kept for the VS Code Copilot Chat picker. - { id: "gpt-5-mini", name: "GPT-5 mini", alias: "gpt-5-mini" }, - { id: "gpt-5.4-nano", name: "GPT-5.4 nano", alias: "gpt-5.4-nano" }, - { id: "claude-haiku-4.5", name: "Claude Haiku 4.5", alias: "claude-haiku-4.5" }, - { id: "gpt-4o", name: "GPT-4o", alias: "gpt-4o" }, - { id: "gpt-4.1", name: "GPT-4.1", alias: "gpt-4.1" }, - ], - }, kiro: { id: "kiro", name: "Kiro", @@ -140,6 +114,34 @@ export const CLI_TOOLS = { description: "OpenAI Codex CLI", configType: "custom", }, + copilot: { + id: "copilot", + name: "GitHub Copilot", + image: "/providers/copilot.png", + color: "#1F6FEB", + description: "GitHub Copilot in VS Code via 9Router extension", + configType: "guide", + docsUrl: "https://marketplace.visualstudio.com/items?itemName=hotrungnhan.9router-for-github-copilot", + guideSteps: [ + { + step: 1, + title: "Install Extension", + desc: "In VS Code, open Extensions (Ctrl+Shift+X or Cmd+Shift+X), search for '9Router for Github Copilot' and click Install.", + }, + { + step: 2, + title: "Configure Server", + desc: "Press Cmd+Shift+P (or Ctrl+Shift+P), run '9Router: Configure Server', then enter your Server URL and API Key:", + value: "{{baseUrl}}", + copyable: true, + }, + { + step: 3, + title: "Select Model in Copilot Chat", + desc: "Open Copilot Chat, click the model picker at the bottom → 'Manage Models...' → check the 9Router models to use.", + }, + ], + }, opencode: { id: "opencode", name: "OpenCode", @@ -197,7 +199,7 @@ export const CLI_TOOLS = { id: "cline", name: "Cline", image: "/providers/cline.png", - color: "#00D1B2", + color: "#5B9BD5", description: "Cline AI Coding Assistant", configType: "custom", }, diff --git a/src/sse/handlers/videoGeneration.js b/src/sse/handlers/videoGeneration.js index 67142899..af19da19 100644 --- a/src/sse/handlers/videoGeneration.js +++ b/src/sse/handlers/videoGeneration.js @@ -5,7 +5,7 @@ import { extractApiKey, isValidApiKey, } from "../services/auth.js"; -import { getSettings } from "@/lib/localDb"; +import { getSettings, getProviderConnectionById } from "@/lib/localDb"; import { getModelInfo } from "../services/model.js"; import { handleVideoProxyCore, getVideoConfig, sanitizeSecrets } from "open-sse/handlers/videoCore.js"; import { errorResponse, unavailableResponse } from "open-sse/utils/error.js"; @@ -17,6 +17,21 @@ import * as log from "../utils/logger.js"; // (bare model id, or multipart bodies we deliberately don't parse) land here. const DEFAULT_VIDEO_PROVIDER = "xai"; +/** + * Poll requests carry no model, so the provider comes from the pinned + * connection (`x-connection-id`, returned on create) or an explicit + * `?provider=` — falling back to the historical xAI default. + */ +async function resolveGetProvider(request, connectionId) { + if (connectionId) { + const conn = await getProviderConnectionById(connectionId).catch(() => null); + if (conn?.provider && getVideoConfig(conn.provider)) return conn.provider; + } + const queried = new URL(request.url).searchParams.get("provider"); + if (queried && getVideoConfig(queried)) return queried; + return DEFAULT_VIDEO_PROVIDER; +} + // Creation POSTs are billable jobs — only rotate to another account for // errors that upstream rejects BEFORE creating a job (auth/quota). A 5xx may // have created the job, so it is returned to the caller instead of re-sent. @@ -185,8 +200,8 @@ export async function handleVideoGet(request, requestId) { if (!requestId) return errorResponse(HTTP_STATUS.BAD_REQUEST, "Missing video request id"); - const provider = DEFAULT_VIDEO_PROVIDER; const preferredConnectionId = request.headers.get("x-connection-id") || null; + const provider = await resolveGetProvider(request, preferredConnectionId); const credentials = await getProviderCredentials(provider, null, null, { preferredConnectionId }); if (!credentials || credentials.allRateLimited) { diff --git a/src/sse/services/auth.js b/src/sse/services/auth.js index ec28b194..fa393b86 100644 --- a/src/sse/services/auth.js +++ b/src/sse/services/auth.js @@ -302,7 +302,7 @@ export async function markAccountUnavailable(connectionId, status, errorText, pr } if (!shouldFallback) return { shouldFallback: false, cooldownMs: 0 }; - const reason = typeof errorText === "string" ? errorText.slice(0, 100) : "Provider error"; + const reason = typeof errorText === "string" ? errorText.slice(0, 200) : "Provider error"; const lockUpdate = buildModelLockUpdate(githubResetAtMs ? null : model, cooldownMs); await updateProviderConnection(connectionId, { diff --git a/src/sse/services/backgroundTokenRefresh.js b/src/sse/services/backgroundTokenRefresh.js index d8bb0a56..db772566 100644 --- a/src/sse/services/backgroundTokenRefresh.js +++ b/src/sse/services/backgroundTokenRefresh.js @@ -9,6 +9,7 @@ import { getCredentialExpiryMs } from "open-sse/services/oauthCredentialManager. export const BACKGROUND_REFRESH_LEAD_MS = 30 * 60 * 1000; const DEFAULT_INTERVAL_MS = 5 * 60 * 1000; const INITIAL_DELAY_MS = 10 * 1000; +const SENSITIVE_PROVIDERS = new Set(["antigravity", "gemini-cli"]); let started = false; let intervalHandle = null; @@ -110,47 +111,47 @@ async function refreshOne(connection) { * @param {{ loadConnections?: Function, refreshConnection?: Function }} [deps] */ export async function runBackgroundTokenRefreshTick(deps = {}) { - if (tickRunning) { - log.debug("BG_TOKEN_REFRESH", "Tick already running, skip"); - return; - } + if (tickRunning) return; tickRunning = true; try { const load = deps.loadConnections || loadActiveConnections; const refresh = deps.refreshConnection || refreshOne; + const sleep = deps.sleep || ((ms) => new Promise((res) => setTimeout(res, ms))); const connections = await load(); const due = selectConnectionsNeedingRefresh(connections, Date.now()); - if (due.length === 0) { - log.debug("BG_TOKEN_REFRESH", "No connections due for refresh", { - active: Array.isArray(connections) ? connections.length : 0, - }); - return; + if (due.length === 0) return; + + const baseSensitiveDelay = Number(process.env.BG_REFRESH_GOOGLE_DELAY_MS) || 12_000; + const baseNormalDelay = Number(process.env.BG_REFRESH_DELAY_MS) || 1_500; + + for (let i = 0; i < due.length; i++) { + const conn = due[i]; + try { + await refresh(conn); + log.info("BG_TOKEN_REFRESH", "Connection refresh finished", { + id: conn.id, + email: conn.email || conn.name || conn.id, + provider: conn.provider, + }); + } catch (err) { + log.warn("BG_TOKEN_REFRESH", "Connection refresh failed (swallowed)", { + id: conn?.id, + email: conn?.email || conn?.name || conn?.id, + provider: conn?.provider, + error: err?.message ?? String(err), + }); + } + + // Sequential delay between accounts to prevent bursting upstream providers (especially Google Cloud) + if (i < due.length - 1) { + const isSensitive = SENSITIVE_PROVIDERS.has(conn.provider); + const baseDelay = isSensitive ? baseSensitiveDelay : baseNormalDelay; + const jitter = isSensitive ? Math.floor(Math.random() * 4000) : 200; + await sleep(baseDelay + jitter); + } } - - log.info("BG_TOKEN_REFRESH", "Refreshing due OAuth connections", { - due: due.length, - ids: due.map((c) => c.id).filter(Boolean), - }); - - await Promise.allSettled( - due.map(async (conn) => { - try { - await refresh(conn); - log.info("BG_TOKEN_REFRESH", "Connection refresh finished", { - id: conn.id, - provider: conn.provider, - }); - } catch (err) { - log.warn("BG_TOKEN_REFRESH", "Connection refresh failed (swallowed)", { - id: conn?.id, - provider: conn?.provider, - error: err?.message ?? String(err), - }); - } - }) - ); } catch (err) { log.warn("BG_TOKEN_REFRESH", "Tick failed (swallowed)", { error: err?.message ?? String(err), @@ -167,14 +168,8 @@ export async function runBackgroundTokenRefreshTick(deps = {}) { */ export function startBackgroundTokenRefresh({ intervalMs } = {}) { if (started) return false; - if (isTruthyEnv(process.env.DISABLE_BACKGROUND_TOKEN_REFRESH)) { - log.info("BG_TOKEN_REFRESH", "Disabled via DISABLE_BACKGROUND_TOKEN_REFRESH"); - return false; - } - if (isNonServerRuntime()) { - log.debug("BG_TOKEN_REFRESH", "Skip start outside long-running server runtime"); - return false; - } + if (isTruthyEnv(process.env.DISABLE_BACKGROUND_TOKEN_REFRESH)) return false; + if (isNonServerRuntime()) return false; started = true; const period = Number.isFinite(intervalMs) && intervalMs > 0 ? intervalMs : DEFAULT_INTERVAL_MS; @@ -194,11 +189,6 @@ export function startBackgroundTokenRefresh({ intervalMs } = {}) { intervalHandle = setInterval(safeTick, period); if (intervalHandle.unref) intervalHandle.unref(); - log.info("BG_TOKEN_REFRESH", "Scheduler started", { - intervalMs: period, - initialDelayMs: INITIAL_DELAY_MS, - leadMs: BACKGROUND_REFRESH_LEAD_MS, - }); return true; } @@ -213,6 +203,5 @@ export function stopBackgroundTokenRefresh() { } if (started) { started = false; - log.info("BG_TOKEN_REFRESH", "Scheduler stopped"); } } diff --git a/src/sse/services/tokenRefresh.js b/src/sse/services/tokenRefresh.js index 4ca55920..83a3244e 100644 --- a/src/sse/services/tokenRefresh.js +++ b/src/sse/services/tokenRefresh.js @@ -125,25 +125,30 @@ function needsProjectId(provider) { function _refreshProjectId(provider, connectionId, accessToken) { if (!needsProjectId(provider) || !connectionId || !accessToken) return; - // Evict the stale cached entry so getProjectIdForConnection does a real fetch + // Invalidate the stale cached entry so getProjectIdForConnection does a real fetch invalidateProjectId(connectionId); - getProjectIdForConnection(connectionId, accessToken) - .then((projectId) => { - if (!projectId) return; - updateProviderCredentials(connectionId, { projectId }).catch((err) => { - log.debug("TOKEN_REFRESH", "Failed to persist refreshed projectId", { + // Lazy resolution: Do not eagerly trigger onboardUser during background token refresh. + // Eagerly fetching projectId across multiple accounts simultaneously triggers Google Cloud anti-abuse / rate limits. + // Runtime handlers (e.g. chat handler) will lazily call getProjectIdForConnection() on demand. + if (process.env.EAGER_PROJECT_ID_REFRESH === "true") { + getProjectIdForConnection(connectionId, accessToken, provider) + .then((projectId) => { + if (!projectId) return; + updateProviderCredentials(connectionId, { projectId }).catch((err) => { + log.debug("TOKEN_REFRESH", "Failed to persist refreshed projectId", { + connectionId, + error: err?.message ?? err, + }); + }); + }) + .catch((err) => { + log.debug("TOKEN_REFRESH", "Failed to fetch projectId after token refresh", { connectionId, error: err?.message ?? err, }); }); - }) - .catch((err) => { - log.debug("TOKEN_REFRESH", "Failed to fetch projectId after token refresh", { - connectionId, - error: err?.message ?? err, - }); - }); + } } // ─── Local-specific: persist credentials to localDb ────────────────────────── diff --git a/tests/__baseline__/providers-baseline.json b/tests/__baseline__/providers-baseline.json index 9d9122b6..27a23427 100644 --- a/tests/__baseline__/providers-baseline.json +++ b/tests/__baseline__/providers-baseline.json @@ -44,6 +44,7 @@ }, "usage": { "quotaApiUrl": "https://daily-cloudcode-pa.googleapis.com/v1internal:fetchAvailableModels", + "quotaSummaryApiUrl": "https://daily-cloudcode-pa.googleapis.com/v1internal:retrieveUserQuotaSummary", "loadProjectApiUrl": "https://cloudcode-pa.googleapis.com/v1internal:loadCodeAssist", "tokenUrl": "https://oauth2.googleapis.com/token" }, @@ -131,6 +132,9 @@ "HTTP-Referer": "https://cline.bot", "X-Title": "Cline" }, + "quirks": { + "clineEnvelope": true + }, "tokenUrl": "https://api.cline.bot/api/v1/auth/token", "refreshUrl": "https://api.cline.bot/api/v1/auth/refresh", "auth": { @@ -150,6 +154,9 @@ "HTTP-Referer": "https://cline.bot", "X-Title": "Cline" }, + "quirks": { + "clineEnvelope": true + }, "auth": { "combined": true, "header": "Authorization", @@ -194,9 +201,10 @@ "baseUrl": "https://chatgpt.com/backend-api/codex/responses", "format": "openai-responses", "forceStream": true, + "cliVersion": "0.154.0", "headers": { "originator": "codex_cli_rs", - "User-Agent": "codex_cli_rs/0.136.0" + "User-Agent": "codex_cli_rs/0.154.0" }, "usage": { "url": "https://chatgpt.com/backend-api/wham/usage", @@ -242,6 +250,12 @@ "reasoningInject": { "scope": "all" }, + "quirks": { + "claudeSupportedToolTypes": [ + "web_search_20250305", + "web_search_20260209" + ] + }, "format": "openai", "transports": [ { @@ -595,7 +609,8 @@ "Anthropic-Beta": "claude-code-20250219,interleaved-thinking-2025-05-14" }, "quirks": { - "dropOutputConfig": true + "dropOutputConfig": true, + "requireClaudeToolType": true }, "reasoningInject": { "scope": "all" @@ -646,7 +661,8 @@ "Anthropic-Beta": "claude-code-20250219,interleaved-thinking-2025-05-14" }, "quirks": { - "dropOutputConfig": true + "dropOutputConfig": true, + "requireClaudeToolType": true }, "reasoningInject": { "scope": "all" @@ -733,6 +749,9 @@ "opencode-go": { "baseUrl": "https://opencode.ai/zen/go/v1/chat/completions", "headers": {}, + "usage": { + "url": "https://opencode.ai/zen/go/v1/usage" + }, "format": "openai", "transports": [ { @@ -770,7 +789,13 @@ "headers": { "x-opencode-client": "desktop" }, + "forceStream": true, "noAuth": true, + "quirks": { + "forceAutoToolChoiceModels": [ + "muse-spark-1.3-contributor-free" + ] + }, "format": "openai" }, "openrouter": { @@ -999,6 +1024,7 @@ "HTTP-Referer": "https://endpoint-proxy.local", "X-Title": "Endpoint Proxy" }, + "forceStream": true, "format": "openai" }, "baidu": { diff --git a/tests/translator/__snapshots__/golden-request.test.js.snap b/tests/translator/__snapshots__/golden-request.test.js.snap index a7a0c219..f5a5253b 100644 --- a/tests/translator/__snapshots__/golden-request.test.js.snap +++ b/tests/translator/__snapshots__/golden-request.test.js.snap @@ -239,7 +239,7 @@ exports[`GOLDEN request: OpenAI → Kiro > full body (image base64 + tool_result "userInputMessage": { "content": "[Context: Current time is -continue", +Tool results provided.", "modelId": "claude-sonnet-4.5", "origin": "AI_EDITOR", "userInputMessageContext": { @@ -281,7 +281,11 @@ continue", "history": [ { "userInputMessage": { - "content": "You are helpful. + "content": "[Context: Current time is + + +You are helpful. + What's in this image?", "images": [ diff --git a/tests/translator/agent-client-fixes.test.js b/tests/translator/agent-client-fixes.test.js new file mode 100644 index 00000000..6004db80 --- /dev/null +++ b/tests/translator/agent-client-fixes.test.js @@ -0,0 +1,188 @@ +// Fixes for agent clients (Claude Code) driving non-Anthropic upstreams: +// - tool-result images survive the Claude → OpenAI / Kiro request translation +// - Kiro tool calls stream back under the client's own (unsanitized) names +// - the client's thinking `display` is kept on Claude-format upstreams +import { describe, it, expect } from "vitest"; +import "./registerAll.js"; +import { translateRequest, translateResponse, initState } from "../../open-sse/translator/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { applyThinking } from "../../open-sse/translator/concerns/thinkingUnified.js"; +import { kiroToClaudeResponse } from "../../open-sse/translator/response/kiro-to-claude.js"; +import { kiroToOpenAIResponse } from "../../open-sse/translator/response/kiro-to-openai.js"; +import { selectAnthropicBeta } from "../../open-sse/providers/shared.js"; +import { hoistToolResultImages } from "../../open-sse/translator/formats/claude.js"; +import { openaiToCommandCodeRequest } from "../../open-sse/translator/request/openai-to-commandcode.js"; + +const PNG = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="; + +const screenshotTurn = (extraTools = []) => ({ + tools: [ + { name: "mcp__browser__computer", description: "browser", input_schema: { type: "object", properties: {} } }, + ...extraTools, + ], + messages: [ + { role: "user", content: "take a screenshot" }, + { role: "assistant", content: [{ type: "tool_use", id: "toolu_1", name: "mcp__browser__computer", input: { action: "screenshot" } }] }, + { + role: "user", + content: [{ + type: "tool_result", + tool_use_id: "toolu_1", + content: [ + { type: "text", text: "Successfully captured screenshot (1x1, png)" }, + { type: "image", source: { type: "base64", media_type: "image/png", data: PNG } }, + ], + }], + }, + ], +}); + +describe("tool-result images reach OpenAI-format upstreams", () => { + it("emits the tool message text and a follow-up user message carrying the image", () => { + const out = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "gpt-x", screenshotTurn(), true, null, "openai"); + const toolMsg = out.messages.find((m) => m.role === "tool"); + expect(toolMsg.tool_call_id).toBe("toolu_1"); + expect(toolMsg.content).toBe("Successfully captured screenshot (1x1, png)"); + expect(toolMsg.content).not.toContain(PNG); + const follow = out.messages[out.messages.indexOf(toolMsg) + 1]; + expect(follow.role).toBe("user"); + const image = follow.content.find((p) => p.type === "image_url"); + expect(image.image_url.url).toBe(`data:image/png;base64,${PNG}`); + expect(follow.content.find((p) => p.type === "text").text).toContain("toolu_1"); + }); + + it("does not dump base64 into a tool message that had no text", () => { + const body = screenshotTurn(); + body.messages[2].content[0].content = [{ type: "image", source: { type: "base64", media_type: "image/png", data: PNG } }]; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "gpt-x", body, true, null, "openai"); + const toolMsg = out.messages.find((m) => m.role === "tool"); + expect(toolMsg.content).toBe(""); + expect(out.messages.some((m) => Array.isArray(m.content) && m.content.some((p) => p.type === "image_url"))).toBe(true); + }); + + it("leaves text-only tool results exactly as before", () => { + const body = screenshotTurn(); + body.messages[2].content[0].content = "plain result"; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "gpt-x", body, true, null, "openai"); + const toolMsg = out.messages.find((m) => m.role === "tool"); + expect(toolMsg.content).toBe("plain result"); + expect(out.messages[out.messages.length - 1]).toBe(toolMsg); + }); + + it("forwards tool-result images to Kiro as user images", () => { + const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", screenshotTurn(), true, null, "kiro"); + const json = JSON.stringify(out.conversationState); + expect(json).toContain(PNG); + expect(json).toContain("Successfully captured screenshot"); + }); +}); + +describe("Kiro tool names round-trip", () => { + it("returns the sanitized→original map on the translated body", () => { + const body = screenshotTurn(); + body.tools = [{ name: "mcp.browser.computer", description: "browser", input_schema: { type: "object", properties: {} } }]; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", body, true, null, "kiro"); + expect(out._toolNameMap).toBeInstanceOf(Map); + expect(out._toolNameMap.get("mcp_browser_computer")).toBe("mcp.browser.computer"); + const wire = JSON.parse(JSON.stringify(out.conversationState)); + expect(JSON.stringify(wire)).toContain("mcp_browser_computer"); + expect(JSON.stringify(wire)).not.toContain("mcp.browser.computer"); + }); + + it("omits the map when no name changed", () => { + const body = screenshotTurn(); + body.tools = [{ name: "plain_tool", description: "x", input_schema: { type: "object", properties: {} } }]; + body.messages[1].content[0].name = "plain_tool"; + const out = translateRequest(FORMATS.CLAUDE, FORMATS.KIRO, "claude-sonnet-4.5", body, true, null, "kiro"); + expect(out._toolNameMap).toBeUndefined(); + }); + + it("restores the client name on streamed Claude tool_use blocks", () => { + const state = { ...initState(FORMATS.CLAUDE), toolNameMap: new Map([["mcp_browser_computer", "mcp__browser__computer"]]) }; + const chunk = { + id: "c1", object: "chat.completion.chunk", created: 1, model: "claude-sonnet-4.5", + choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_1", type: "function", function: { name: "mcp_browser_computer", arguments: "" } }] }, finish_reason: null }], + }; + const events = kiroToClaudeResponse(chunk, state); + const start = events.find((e) => e.type === "content_block_start" && e.content_block?.type === "tool_use"); + expect(start.content_block.name).toBe("mcp__browser__computer"); + }); + + it("passes unknown names through untouched", () => { + const state = { ...initState(FORMATS.CLAUDE), toolNameMap: new Map([["mcp_browser_computer", "mcp__browser__computer"]]) }; + const chunk = { + id: "c1", object: "chat.completion.chunk", created: 1, model: "claude-sonnet-4.5", + choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_2", type: "function", function: { name: "other_tool", arguments: "" } }] }, finish_reason: null }], + }; + const events = kiroToClaudeResponse(chunk, state); + const start = events.find((e) => e.type === "content_block_start" && e.content_block?.type === "tool_use"); + expect(start.content_block.name).toBe("other_tool"); + }); + + it("restores the client name on OpenAI chunks passed through kiro-to-openai", () => { + const state = { ...initState(FORMATS.OPENAI), toolNameMap: new Map([["mcp_browser_computer", "mcp__browser__computer"]]) }; + const chunk = { + id: "c1", object: "chat.completion.chunk", created: 1, model: "claude-sonnet-4.5", + choices: [{ index: 0, delta: { tool_calls: [{ index: 0, id: "call_1", type: "function", function: { name: "mcp_browser_computer", arguments: "{}" } }] }, finish_reason: null }], + }; + const out = kiroToOpenAIResponse(chunk, state); + expect(out.choices[0].delta.tool_calls[0].function.name).toBe("mcp__browser__computer"); + expect(kiroToOpenAIResponse(chunk, initState(FORMATS.OPENAI))).toBe(chunk); + }); +}); + +describe("thinking display is preserved for Claude-format upstreams", () => { + it("keeps display on adaptive thinking", () => { + const body = { model: "claude-sonnet-5", thinking: { type: "adaptive", display: "summarized" }, output_config: { effort: "high" }, messages: [] }; + applyThinking(FORMATS.CLAUDE, "claude-sonnet-5", body, "claude"); + expect(body.thinking).toEqual({ type: "adaptive", display: "summarized" }); + expect(body.output_config).toEqual({ effort: "high" }); + }); + + it("keeps display on budget thinking and omits it when the client sent none", () => { + const withDisplay = { model: "claude-haiku-4-5-20251001", thinking: { type: "adaptive", display: "omitted" }, output_config: { effort: "low" }, messages: [] }; + applyThinking(FORMATS.CLAUDE, "claude-haiku-4-5-20251001", withDisplay, "claude"); + expect(withDisplay.thinking.type).toBe("enabled"); + expect(withDisplay.thinking.display).toBe("omitted"); + + const without = { model: "claude-sonnet-5", thinking: { type: "adaptive" }, output_config: { effort: "high" }, messages: [] }; + applyThinking(FORMATS.CLAUDE, "claude-sonnet-5", without, "claude"); + expect(without.thinking).toEqual({ type: "adaptive" }); + }); +}); + +describe("redact-thinking beta follows the client's display request", () => { + it("keeps redact-thinking by default and drops it for summarized display", () => { + expect(selectAnthropicBeta("claude-sonnet-5")).toContain("redact-thinking-2026-02-12"); + expect(selectAnthropicBeta("claude-sonnet-5", { thinking: { type: "adaptive", display: "omitted" } })).toContain("redact-thinking-2026-02-12"); + const summarized = selectAnthropicBeta("claude-sonnet-5", { thinking: { type: "adaptive", display: "summarized" } }); + expect(summarized).not.toContain("redact-thinking-2026-02-12"); + expect(summarized).toContain("interleaved-thinking-2025-05-14"); + expect(summarized).toContain("effort-2025-11-24"); + }); +}); + +describe("tool-result images reach Anthropic-compatible and Command Code upstreams", () => { + it("hoists a tool_result image into the same user turn after the results", () => { + const body = screenshotTurn(); + const out = hoistToolResultImages(body); + const user = out.messages[2]; + expect(user.content[0].type).toBe("tool_result"); + expect(user.content[0].content.every((c) => c.type !== "image")).toBe(true); + expect(user.content.some((c) => c.type === "image" && c.source?.data === PNG)).toBe(true); + expect(user.content.find((c) => c.type === "text" && /toolu_1/.test(c.text))).toBeTruthy(); + // No image: untouched object identity. + const plain = { messages: [{ role: "user", content: [{ type: "tool_result", tool_use_id: "x", content: "ok" }] }] }; + expect(hoistToolResultImages(plain)).toBe(plain); + }); + + it("sends an image block to Command Code instead of a placeholder", () => { + const openaiBody = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, "muse-spark", screenshotTurn(), true, null, "commandcode"); + const out = openaiToCommandCodeRequest("muse-spark", openaiBody, true); + const json = JSON.stringify(out); + expect(json).not.toContain("[image omitted]"); + expect(json).toContain(`"type":"image"`); + expect(json).toContain(`data:image/png;base64,${PNG}`); + expect(json).toContain(`"mediaType":"image/png"`); + }); +}); diff --git a/tests/translator/bugs-3905-deepseek-tool-type.test.js b/tests/translator/bugs-3905-deepseek-tool-type.test.js new file mode 100644 index 00000000..fd561d0b --- /dev/null +++ b/tests/translator/bugs-3905-deepseek-tool-type.test.js @@ -0,0 +1,33 @@ +// Regression for #3905: defaultClaudeToolType() (type:"custom") must only run for +// gateways that declare the requireClaudeToolType quirk (MiniMax). Claude-format +// endpoints that only accept the legacy typeless tool shape — e.g. DeepSeek's +// Anthropic-compatible endpoint, which answers HTTP 400 "unknown variant `custom`" — +// must never receive tools[].type = "custom". +import { describe, it, expect } from "vitest"; +import { PROVIDERS } from "../../open-sse/providers/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import { shouldDefaultClaudeToolType } from "../../open-sse/translator/concerns/toolCall.js"; + +const tools = [{ name: "get_weather", description: "weather", input_schema: { type: "object" } }]; + +describe("Claude tool `type` defaulting is provider-scoped (#3905)", () => { + it("runs only for providers declaring requireClaudeToolType", () => { + expect(shouldDefaultClaudeToolType("minimax", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(true); + expect(shouldDefaultClaudeToolType("minimax-cn", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(true); + // Endpoints accepting only the legacy typeless shape must NOT get type:"custom". + expect(shouldDefaultClaudeToolType("deepseek", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(false); + expect(shouldDefaultClaudeToolType("claude", FORMATS.CLAUDE, tools, PROVIDERS)).toBe(false); + }); + + it("never applies outside Claude-format requests or without tools", () => { + expect(shouldDefaultClaudeToolType("minimax", FORMATS.OPENAI, tools, PROVIDERS)).toBe(false); + expect(shouldDefaultClaudeToolType("minimax", FORMATS.CLAUDE, undefined, PROVIDERS)).toBe(false); + expect(shouldDefaultClaudeToolType("minimax", FORMATS.CLAUDE, null, PROVIDERS)).toBe(false); + }); + + it("declares the quirk only on the MiniMax providers (registry tripwire)", () => { + expect(PROVIDERS.minimax?.quirks?.requireClaudeToolType).toBe(true); + expect(PROVIDERS["minimax-cn"]?.quirks?.requireClaudeToolType).toBe(true); + expect(PROVIDERS.deepseek?.quirks?.requireClaudeToolType).toBeUndefined(); + }); +}); diff --git a/tests/translator/bugs-gemini-cursor-commandcode.test.js b/tests/translator/bugs-gemini-cursor-commandcode.test.js index 4b74c60e..8483bfc1 100644 --- a/tests/translator/bugs-gemini-cursor-commandcode.test.js +++ b/tests/translator/bugs-gemini-cursor-commandcode.test.js @@ -62,15 +62,18 @@ describe("OpenAI → CommandCode", () => { expect(Object.keys(call.input).length, "arguments silently dropped to {}").toBeGreaterThan(0); }); - // openai-to-commandcode.js:41-42 — image becomes "[image omitted]" - // KNOWN BUG - it.fails("image content is preserved", () => { + it("image content is preserved as native CommandCode image blocks", () => { const out = O2CC({ messages: [{ role: "user", content: [ { type: "text", text: "look" }, { type: "image_url", image_url: { url: "data:image/png;base64,BBBB" } }, ] }], }); - expect(JSON.stringify(out), "image omitted").toContain("BBBB"); + expect(JSON.stringify(out)).toContain("BBBB"); + expect(JSON.stringify(out)).not.toContain("[image omitted]"); + expect(out.params.messages[0].content).toEqual([ + { type: "text", text: "look" }, + { type: "image", image: "data:image/png;base64,BBBB", mimeType: "image/png" }, + ]); }); }); diff --git a/tests/translator/claude-kiro-direct.test.js b/tests/translator/claude-kiro-direct.test.js index 3fca7be1..ab122b0c 100644 --- a/tests/translator/claude-kiro-direct.test.js +++ b/tests/translator/claude-kiro-direct.test.js @@ -26,9 +26,8 @@ describe("Claude → Kiro (direct route)", () => { expect(first.conversationState.conversationId).toBe("hermes-session-123-claude-replay"); expect(second.conversationState.conversationId).toBe("hermes-session-123-claude-replay"); - expect(first.conversationState.agentContinuationId).toBeTruthy(); - expect(second.conversationState.agentContinuationId).toBe(first.conversationState.agentContinuationId); - expect(first.conversationState.agentTaskType).toBe("vibe"); + expect(first.conversationState).not.toHaveProperty("agentContinuationId"); + expect(second.conversationState).not.toHaveProperty("agentTaskType"); expect(second.conversationState.history[0].userInputMessage.content).toBe( first.conversationState.currentMessage.userInputMessage.content ); @@ -84,7 +83,7 @@ describe("Claude → Kiro (direct route)", () => { expect(out.systemPrompt).toContain( "enabled" ); - expect(out.agentMode).toBe("vibe"); + expect(out).not.toHaveProperty("agentMode"); }); it("does not send additionalModelRequestFields for Kiro models without effort support", () => { diff --git a/tests/translator/kiro-tool-name-roundtrip.test.js b/tests/translator/kiro-tool-name-roundtrip.test.js new file mode 100644 index 00000000..a8f151ad --- /dev/null +++ b/tests/translator/kiro-tool-name-roundtrip.test.js @@ -0,0 +1,110 @@ +import { describe, it, expect } from "vitest"; +import { normalizeKiroToolSpecs } from "../../open-sse/translator/concerns/kiroConversation.js"; +import { openaiToKiroRequest } from "../../open-sse/translator/request/openai-to-kiro.js"; +import { claudeToKiroRequest } from "../../open-sse/translator/request/claude-to-kiro.js"; +import { kiroToOpenAIResponse } from "../../open-sse/translator/response/kiro-to-openai.js"; +import { kiroToClaudeResponse, kiroToClaudeNonStreaming } from "../../open-sse/translator/response/kiro-to-claude.js"; + +describe("Kiro tool name normalization and roundtrip", () => { + it("preserves consecutive underscores like mcp__gitea__search_repos without collapsing", () => { + const { specs, nameMap } = normalizeKiroToolSpecs([ + { name: "mcp__gitea__search_repos", description: "Search Gitea" }, + ]); + expect(specs).toHaveLength(1); + expect(specs[0].toolSpecification.name).toBe("mcp__gitea__search_repos"); + expect(nameMap.get("mcp__gitea__search_repos")).toBe("mcp__gitea__search_repos"); + }); + + it("builds _toolNameMap for illegal characters and deduplicates colliding names", () => { + const tools = [ + { name: "my.tool/search", description: "tool 1" }, + { name: "my_tool_search", description: "tool 2" }, + ]; + const openaiPayload = openaiToKiroRequest("claude-sonnet-4.6", { + tools: tools.map((t) => ({ type: "function", function: t })), + messages: [{ role: "user", content: "hello" }], + }, true, {}); + + expect(openaiPayload._toolNameMap).toBeInstanceOf(Map); + // my.tool/search cleaned to my_tool_search. Since my_tool_search comes next, it becomes my_tool_search_2 + expect(openaiPayload._toolNameMap.get("my_tool_search")).toBe("my.tool/search"); + + const claudePayload = claudeToKiroRequest("claude-sonnet-4.6", { + tools, + messages: [{ role: "user", content: "hello" }], + }, true, {}); + + expect(claudePayload._toolNameMap).toBeInstanceOf(Map); + expect(claudePayload._toolNameMap.get("my_tool_search")).toBe("my.tool/search"); + }); + + it("does not attach _toolNameMap when all tool names are legal and unchanged", () => { + const tools = [ + { name: "mcp__gitea__search_repos", description: "Search Gitea" }, + { name: "bash_exec", description: "Run bash" }, + ]; + const payload = openaiToKiroRequest("claude-sonnet-4.6", { + tools: tools.map((t) => ({ type: "function", function: t })), + messages: [{ role: "user", content: "hello" }], + }, true, {}); + + expect(payload._toolNameMap).toBeUndefined(); + }); + + it("restores original tool name in kiroToOpenAIResponse when state.toolNameMap is present", () => { + const state = { + toolNameMap: new Map([["my_tool_search", "my.tool/search"]]), + }; + const event = { + toolUseEvent: { + toolUseId: "call_123", + name: "my_tool_search", + input: { q: "test" }, + }, + }; + const chunk = kiroToOpenAIResponse(event, state); + expect(chunk).not.toBeNull(); + expect(chunk.choices[0].delta.tool_calls[0].function.name).toBe("my.tool/search"); + }); + + it("restores original tool name in kiroToClaudeResponse streaming when state.toolNameMap is present", () => { + const state = { + toolNameMap: new Map([["my_tool_search", "my.tool/search"]]), + toolCalls: new Map(), + nextBlockIndex: 0, + }; + const chunk = { + id: "chatcmpl-1", + choices: [{ + delta: { + tool_calls: [{ + index: 0, + id: "call_123", + type: "function", + function: { name: "my_tool_search", arguments: "" }, + }], + }, + }], + }; + const events = kiroToClaudeResponse(chunk, state); + const startEvent = events.find((e) => e.type === "content_block_start"); + expect(startEvent).toBeDefined(); + expect(startEvent.content_block.name).toBe("my.tool/search"); + }); + + it("restores original tool name in kiroToClaudeNonStreaming when toolNameMap is present", () => { + const data = { + choices: [{ + message: { + tool_calls: [{ + id: "call_123", + function: { name: "my_tool_search", arguments: "{}" }, + }], + }, + }], + toolNameMap: new Map([["my_tool_search", "my.tool/search"]]), + }; + const result = kiroToClaudeNonStreaming(data); + expect(result.content[0].name).toBe("my.tool/search"); + }); +}); diff --git a/tests/translator/thinking-unified.test.js b/tests/translator/thinking-unified.test.js index 93d8b683..f04b18bd 100644 --- a/tests/translator/thinking-unified.test.js +++ b/tests/translator/thinking-unified.test.js @@ -82,6 +82,16 @@ describe("applyThinking per provider format", () => { // Sonnet 5). Both fields together are the documented adaptive shape. expect(out.thinking).toEqual({ type: "adaptive" }); }); + it("claude adaptive thinking maps auto effort to a supported level", () => { + const out = apply("claude", "claude-opus-4.7", { thinking: { type: "adaptive" } }, "claude"); + expect(out.output_config).toEqual({ effort: "high" }); + expect(out.thinking).toEqual({ type: "adaptive" }); + }); + it("permanently adaptive Claude maps auto effort without adding a thinking switch", () => { + const out = apply("claude", "claude-fable-5-1", { thinking: { type: "adaptive" } }, "claude"); + expect(out.output_config).toEqual({ effort: "high" }); + expect(out.thinking).toBeUndefined(); + }); it("Fable 5.1 → effort without a redundant thinking switch", () => { const out = apply("claude", "claude-fable-5-1", { reasoning_effort: "high" }, "claude"); expect(out.output_config).toEqual({ effort: "high" }); @@ -240,6 +250,29 @@ describe("applyThinking per provider format", () => { const out = apply("gemini-cli", "gemini-3.5-flash-lite", { reasoning_effort: "medium" }, "gemini-cli"); expect(out.generationConfig.thinkingConfig.thinkingLevel).toBe("medium"); }); + it("commandcode envelope writes params.reasoning_effort, not wrapper fields", () => { + const out = apply("commandcode", "deepseek/deepseek-v4.1-flash", { + params: { model: "deepseek/deepseek-v4.1-flash", messages: [] }, + reasoning_effort: "high", + }, "commandcode"); + expect(out.params.reasoning_effort).toBe("high"); + expect(out.reasoning_effort).toBeUndefined(); + expect(out.thinking).toBeUndefined(); + }); + it("commandcode preserves low effort instead of remapping to high", () => { + const out = apply("commandcode", "deepseek/deepseek-v4.1-flash", { + params: { messages: [] }, + reasoning_effort: "low", + }, "commandcode"); + expect(out.params.reasoning_effort).toBe("low"); + }); + it("commandcode preserves max effort", () => { + const out = apply("commandcode", "deepseek/deepseek-v4.1-flash", { + params: { messages: [] }, + reasoning_effort: "max", + }, "commandcode"); + expect(out.params.reasoning_effort).toBe("max"); + }); }); describe("extractReasoningText (response shapes)", () => { diff --git a/tests/unit/account-fallback-4xx.test.js b/tests/unit/account-fallback-4xx.test.js new file mode 100644 index 00000000..ba94c16c --- /dev/null +++ b/tests/unit/account-fallback-4xx.test.js @@ -0,0 +1,38 @@ +// Regression: an unmatched 4xx (a request-scoped failure) used to hit the +// transient-cooldown default, which locked the account for 30s and — with a +// single connection — answered every other request in that window with a copy of +// the first error. A 400 "maximum context length" from one session therefore +// looked like the same failure in unrelated sessions. +import { describe, expect, it } from "vitest"; +import { checkFallbackError } from "../../open-sse/services/accountFallback.js"; + +describe("checkFallbackError — request-scoped vs account-scoped failures", () => { + it("does not cool the account down for a 400 caused by the request", () => { + const result = checkFallbackError(400, JSON.stringify({ + error: { + message: "This model's maximum context length is 1048576 tokens. However, you requested 1186139 tokens", + type: "invalid_request_error", + }, + })); + + expect(result).toEqual({ shouldFallback: false, cooldownMs: 0 }); + }); + + it("still falls back for account-scoped statuses", () => { + for (const status of [401, 402, 403, 404, 429]) { + expect(checkFallbackError(status, "nope").shouldFallback).toBe(true); + } + }); + + it("still honours rate-limit / quota wording on any 4xx", () => { + expect(checkFallbackError(400, "rate limit reached").shouldFallback).toBe(true); + expect(checkFallbackError(422, "quota exceeded").shouldFallback).toBe(true); + }); + + it("keeps the transient cooldown for unmatched server errors", () => { + const result = checkFallbackError(503, "upstream exploded"); + + expect(result.shouldFallback).toBe(true); + expect(result.cooldownMs).toBeGreaterThan(0); + }); +}); diff --git a/tests/unit/antigravity-billing-header-rewrite.test.js b/tests/unit/antigravity-billing-header-rewrite.test.js new file mode 100644 index 00000000..e324294f --- /dev/null +++ b/tests/unit/antigravity-billing-header-rewrite.test.js @@ -0,0 +1,38 @@ +import { describe, expect, it } from "vitest"; + +import { AntigravityExecutor } from "../../open-sse/executors/antigravity.js"; +import { openaiToAntigravityRequest } from "../../open-sse/translator/request/openai-to-gemini.js"; + +const HEADER = "x-anthropic-billing-header: cc_version=2.1.275.f15; cc_entrypoint=cli;"; + +function systemTextSentToAntigravity(systemContent) { + // OpenAI-format client (e.g. a proxy converting Claude Code to /v1/chat/completions). + const body = openaiToAntigravityRequest("gemini-3.8-flash-tiered", { + messages: [ + { role: "system", content: systemContent }, + { role: "user", content: "hi" }, + ], + }, true); + const finalBody = new AntigravityExecutor().transformRequest("gemini-3.8-flash-tiered", body, true, {}); + return finalBody.request.systemInstruction.parts.map((p) => p.text).join("\n"); +} + +describe("Antigravity strips the Claude Code billing header from system prompts", () => { + it("removes the header line prepended by Claude Code", () => { + const text = systemTextSentToAntigravity(`${HEADER}\n\nYou are Claude Code, Anthropic's official CLI for Claude.`); + expect(text).not.toContain("x-anthropic-billing-header"); + expect(text).toContain("You are Claude Code, Anthropic's official CLI for Claude."); + }); + + it("removes the header when it is not the first line", () => { + const text = systemTextSentToAntigravity(`Some preamble\n${HEADER}\nRest of prompt`); + expect(text).not.toContain("x-anthropic-billing-header"); + expect(text).toContain("Some preamble"); + expect(text).toContain("Rest of prompt"); + }); + + it("leaves prompts without the header untouched", () => { + const text = systemTextSentToAntigravity("You are a helpful assistant."); + expect(text).toContain("You are a helpful assistant."); + }); +}); diff --git a/tests/unit/antigravity-quota-gemini-3.6.test.js b/tests/unit/antigravity-quota-gemini-3.6.test.js index c67924d6..042983d1 100644 --- a/tests/unit/antigravity-quota-gemini-3.6.test.js +++ b/tests/unit/antigravity-quota-gemini-3.6.test.js @@ -4,7 +4,7 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } : { models: { "gemini-3.6-flash-high": { diff --git a/tests/unit/antigravity-quota-gemini-3.7.test.js b/tests/unit/antigravity-quota-gemini-3.7.test.js index e172be31..e1cad3a9 100644 --- a/tests/unit/antigravity-quota-gemini-3.7.test.js +++ b/tests/unit/antigravity-quota-gemini-3.7.test.js @@ -4,7 +4,7 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } : { models: { "gemini-3.7-flash-high": { diff --git a/tests/unit/antigravity-quota-gemini-3.8.test.js b/tests/unit/antigravity-quota-gemini-3.8.test.js index cb8637e5..8dd0056c 100644 --- a/tests/unit/antigravity-quota-gemini-3.8.test.js +++ b/tests/unit/antigravity-quota-gemini-3.8.test.js @@ -4,7 +4,7 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } : { models: { "gemini-3.8-flash-high": { diff --git a/tests/unit/antigravity-thought-signature-family.test.js b/tests/unit/antigravity-thought-signature-family.test.js new file mode 100644 index 00000000..7e34a17c --- /dev/null +++ b/tests/unit/antigravity-thought-signature-family.test.js @@ -0,0 +1,104 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +// Keep the signature store in RAM only; the SQLite kv layer is not under test here. +vi.mock("@/lib/db/helpers/kvStore.js", () => ({ + makeKv: () => ({ + get: async () => null, + set: async () => {}, + remove: async () => {}, + getAll: async () => ({}), + }), +})); + +const { + storeGeminiThoughtSignature, + getGeminiThoughtSignatureSync, + signatureFamily, +} = await import("../../open-sse/services/thoughtSignatureStore.js"); +const { openaiToAntigravityRequest } = await import("../../open-sse/translator/request/openai-to-gemini.js"); +const { geminiToOpenAIResponse } = await import("../../open-sse/translator/response/gemini-to-openai.js"); +const { DEFAULT_THINKING_GEMINI_CLI_SIGNATURE } = await import("../../open-sse/config/defaultThinkingSignature.js"); + +let n = 0; +const uid = (p) => `${p}_${Date.now()}_${n++}`; + +function toolHistory(callId) { + return { + messages: [ + { role: "user", content: "list files" }, + { role: "assistant", content: null, tool_calls: [{ id: callId, type: "function", function: { name: "ls", arguments: "{}" } }] }, + { role: "tool", tool_call_id: callId, content: "a.txt" }, + ], + }; +} + +function functionCallSignatures(req) { + return req.request.contents.flatMap((c) => c.parts || []).filter((p) => p.functionCall).map((p) => p.thoughtSignature); +} + +describe("antigravity thought signatures are scoped to the model family", () => { + beforeEach(() => { n++; }); + + it("classifies model families", () => { + expect(signatureFamily("claude-opus-4-6-thinking")).toBe("claude"); + expect(signatureFamily("gemini-3.8-flash-tiered")).toBe("gemini"); + expect(signatureFamily("gpt-oss-120b-medium")).toBe("gpt-oss-120b-medium"); + expect(signatureFamily(null)).toBe(null); + }); + + it("does not return a Claude signature for a Gemini target (and vice versa)", () => { + const claudeCall = uid("toolu"); + const geminiCall = uid("call"); + storeGeminiThoughtSignature(claudeCall, "CLAUDE_SIG", "sess", "claude-opus-4-6-thinking"); + storeGeminiThoughtSignature(geminiCall, "GEMINI_SIG", "sess", "gemini-3.8-flash-tiered"); + + expect(getGeminiThoughtSignatureSync(claudeCall, "sess", "gemini-3.8-flash")).toBe(null); + expect(getGeminiThoughtSignatureSync(claudeCall, "sess", "claude-opus-4-6-thinking")).toBe("CLAUDE_SIG"); + expect(getGeminiThoughtSignatureSync(geminiCall, "sess", "gemini-3.7-flash")).toBe("GEMINI_SIG"); + expect(getGeminiThoughtSignatureSync(geminiCall, null, "claude-sonnet-4-6")).toBe(null); + }); + + it("keeps old behaviour for untagged entries and untargeted lookups", () => { + const call = uid("call"); + storeGeminiThoughtSignature(call, "LEGACY_SIG", "sess"); + expect(getGeminiThoughtSignatureSync(call, "sess", "gemini-3.8-flash")).toBe("LEGACY_SIG"); + + const tagged = uid("toolu"); + storeGeminiThoughtSignature(tagged, "CLAUDE_SIG", "sess", "claude-opus-4-6-thinking"); + expect(getGeminiThoughtSignatureSync(tagged, "sess")).toBe("CLAUDE_SIG"); + }); + + it("records the producing model from the Gemini response stream", () => { + const call = uid("toolu_vrtx"); + const state = { model: "claude-opus-4-6-thinking", sessionId: null, toolNameMap: null }; + geminiToOpenAIResponse({ + response: { + responseId: "r1", + candidates: [{ content: { role: "model", parts: [{ functionCall: { id: call, name: "ls", args: {} }, thoughtSignature: "CLAUDE_SIG" }] } }], + }, + }, state); + expect(getGeminiThoughtSignatureSync(call, null, "gemini-3.8-flash-tiered")).toBe(null); + expect(getGeminiThoughtSignatureSync(call, null, "claude-opus-4-6-thinking")).toBe("CLAUDE_SIG"); + }); + + it("switching Claude -> Gemini mid-conversation sends the default signature, not Claude's", () => { + const call = uid("toolu_vrtx"); + storeGeminiThoughtSignature(call, "CLAUDE_SIG", null, "claude-opus-4-6-thinking"); + const req = openaiToAntigravityRequest("gemini-3.8-flash-tiered", toolHistory(call), true); + expect(functionCallSignatures(req)).toEqual([DEFAULT_THINKING_GEMINI_CLI_SIGNATURE]); + }); + + it("switching Gemini -> Claude mid-conversation does not replay Gemini's signature", () => { + const call = uid("call"); + storeGeminiThoughtSignature(call, "GEMINI_SIG", null, "gemini-3.8-flash-tiered"); + const req = openaiToAntigravityRequest("claude-opus-4-6-thinking", toolHistory(call), true); + expect(functionCallSignatures(req)).not.toContain("GEMINI_SIG"); + }); + + it("same model family still reuses the cached signature", () => { + const call = uid("call"); + storeGeminiThoughtSignature(call, "GEMINI_SIG", null, "gemini-3.8-flash-tiered"); + const req = openaiToAntigravityRequest("gemini-3.8-flash-tiered", toolHistory(call), true); + expect(functionCallSignatures(req)).toEqual(["GEMINI_SIG"]); + }); +}); diff --git a/tests/unit/antigravity-usage-headers.test.js b/tests/unit/antigravity-usage-headers.test.js index 74cf5287..363f5b6c 100644 --- a/tests/unit/antigravity-usage-headers.test.js +++ b/tests/unit/antigravity-usage-headers.test.js @@ -4,8 +4,10 @@ const proxyAwareFetch = vi.fn(async (url) => ({ ok: true, status: 200, json: async () => url.includes(":loadCodeAssist") - ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" } } - : { models: {} }, + ? { cloudaicompanionProject: "project-1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } } + : url.includes(":retrieveUserQuotaSummary") + ? { groups: [] } + : { models: {} }, text: async () => "{}", })); @@ -21,7 +23,8 @@ describe("Antigravity usage headers", () => { await getAntigravityUsage("access-token", {}); - expect(proxyAwareFetch).toHaveBeenCalledTimes(2); + // loadCodeAssist + fetchAvailableModels + retrieveUserQuotaSummary + expect(proxyAwareFetch).toHaveBeenCalledTimes(3); for (const [, options] of proxyAwareFetch.mock.calls) { expect(options.headers["User-Agent"]).toBe("antigravity/ide/2.11.0 darwin/arm64"); expect(options.headers).not.toHaveProperty("x-request-source"); diff --git a/tests/unit/antigravity-weekly-dashboard.test.js b/tests/unit/antigravity-weekly-dashboard.test.js new file mode 100644 index 00000000..1c4a2784 --- /dev/null +++ b/tests/unit/antigravity-weekly-dashboard.test.js @@ -0,0 +1,113 @@ +import { describe, it, expect } from "vitest"; +import { parseQuotaData } from "@/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js"; + +describe("Antigravity dashboard normalization with weekly quotas", () => { + const data = { + quotas: { + "gemini-pro-agent": { + displayName: "Gemini 3.1 Pro (High)", + used: 200, + total: 1000, + resetAt: "2026-09-08T00:00:00Z", + remainingPercentage: 80, + }, + "claude-opus-4-6-thinking": { + displayName: "Claude Opus 4.6 (Thinking)", + used: 100, + total: 1000, + resetAt: "2026-09-08T00:00:00Z", + remainingPercentage: 90, + }, + gemini_weekly: { + displayName: "Gemini (Weekly)", + used: 250, + total: 1000, + resetAt: "2026-09-15T00:00:00Z", + remainingPercentage: 75, + }, + claude_gpt_weekly: { + displayName: "Claude & GPT (Weekly)", + used: 500, + total: 1000, + resetAt: "2026-09-14T00:00:00Z", + remainingPercentage: 50, + }, + }, + }; + + it("includes weekly rows with correct display names", () => { + const quotas = parseQuotaData("antigravity", data); + const names = quotas.map((q) => q.name); + + expect(names).toContain("Gemini (Flash / Pro)"); + expect(names).toContain("Claude (Sonnet / Opus)"); + expect(names).toContain("Gemini (Weekly)"); + expect(names).toContain("Claude & GPT (Weekly)"); + }); + + it("uses stable modelKey for weekly rows", () => { + const quotas = parseQuotaData("antigravity", data); + const keys = quotas.map((q) => q.modelKey); + + expect(keys).toContain("gemini_weekly"); + expect(keys).toContain("claude_gpt_weekly"); + }); + + it("weekly rows carry correct quota values", () => { + const quotas = parseQuotaData("antigravity", data); + const geminiWeekly = quotas.find((q) => q.modelKey === "gemini_weekly"); + const claudeWeekly = quotas.find((q) => q.modelKey === "claude_gpt_weekly"); + + expect(geminiWeekly).toMatchObject({ + used: 250, + total: 1000, + remainingPercentage: 75, + resetAt: "2026-09-15T00:00:00Z", + }); + expect(claudeWeekly).toMatchObject({ + used: 500, + total: 1000, + remainingPercentage: 50, + resetAt: "2026-09-14T00:00:00Z", + }); + }); + + it("weekly rows do NOT appear as otherModels", () => { + const quotas = parseQuotaData("antigravity", data); + const weeklyRows = quotas.filter((q) => + q.modelKey === "gemini_weekly" || q.modelKey === "claude_gpt_weekly" + ); + expect(weeklyRows).toHaveLength(2); + expect(weeklyRows[0].name).toMatch(/Weekly/); + expect(weeklyRows[1].name).toMatch(/Weekly/); + }); + + it("order: gemini family, claude family, weekly, then other", () => { + const quotas = parseQuotaData("antigravity", data); + const keys = quotas.map((q) => q.modelKey); + + const geminiIdx = keys.indexOf("gemini"); + const claudeIdx = keys.indexOf("claude"); + const geminiWeeklyIdx = keys.indexOf("gemini_weekly"); + const claudeWeeklyIdx = keys.indexOf("claude_gpt_weekly"); + + expect(geminiIdx).toBeLessThan(geminiWeeklyIdx); + expect(claudeIdx).toBeLessThan(claudeWeeklyIdx); + }); + + it("works with no weekly keys present (backward compat)", () => { + const noWeekly = { + quotas: { + "gemini-pro-agent": { + displayName: "Gemini 3.1 Pro (High)", + used: 200, + total: 1000, + remainingPercentage: 80, + }, + }, + }; + const quotas = parseQuotaData("antigravity", noWeekly); + expect(quotas).toHaveLength(1); + expect(quotas[0].name).toBe("Gemini (Flash / Pro)"); + }); +}); diff --git a/tests/unit/antigravity-weekly-quota.test.js b/tests/unit/antigravity-weekly-quota.test.js new file mode 100644 index 00000000..55cb3a81 --- /dev/null +++ b/tests/unit/antigravity-weekly-quota.test.js @@ -0,0 +1,464 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +// Mock proxyAwareFetch before any imports that use it +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(), +})); + +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; +import { + parseWeeklyQuotaSummary, + fetchAntigravityWeeklyQuota, + _clearWeeklyCache, +} from "../../open-sse/services/usage/antigravity-weekly.js"; + +// — Fixtures —————————————————————————————————————————————— +const GEMINI_GROUP = { + displayName: "Gemini Models", + buckets: [ + { + bucketId: "gemini-weekly-bucket", + displayName: "Weekly Limit", + remainingFraction: 0.75, + resetTime: "2026-09-15T00:00:00Z", + }, + { + bucketId: "gemini-daily-bucket", + displayName: "Daily Limit", + remainingFraction: 0.9, + resetTime: "2026-09-09T00:00:00Z", + }, + ], +}; + +const CLAUDE_GPT_GROUP = { + displayName: "Claude and GPT models", + buckets: [ + { + bucketId: "claude-gpt-weekly", + displayName: "Weekly Quota", + remainingFraction: 0.5, + resetTime: "2026-09-14T00:00:00Z", + }, + ], +}; + +const FULL_RESPONSE = { groups: [GEMINI_GROUP, CLAUDE_GPT_GROUP] }; + +const NESTED_RESPONSE = { + quotaSummary: { + groups: [GEMINI_GROUP, CLAUDE_GPT_GROUP], + }, +}; + +// — parseWeeklyQuotaSummary ——————————————————————————————— +describe("parseWeeklyQuotaSummary", () => { + it("extracts Gemini weekly quota from top-level groups", () => { + const result = parseWeeklyQuotaSummary(FULL_RESPONSE); + expect(result.gemini_weekly).toMatchObject({ + used: 250, + total: 1000, + remainingPercentage: 75, + displayName: "Gemini (Weekly)", + unlimited: false, + }); + expect(result.gemini_weekly.resetAt).toBe("2026-09-15T00:00:00.000Z"); + }); + + it("extracts Claude & GPT weekly quota", () => { + const result = parseWeeklyQuotaSummary(FULL_RESPONSE); + expect(result.claude_gpt_weekly).toMatchObject({ + used: 500, + total: 1000, + remainingPercentage: 50, + displayName: "Claude & GPT (Weekly)", + unlimited: false, + }); + expect(result.claude_gpt_weekly.resetAt).toBe("2026-09-14T00:00:00.000Z"); + }); + + it("handles alternate nested quotaSummary.groups shape", () => { + const result = parseWeeklyQuotaSummary(NESTED_RESPONSE); + expect(result.gemini_weekly).toBeDefined(); + expect(result.claude_gpt_weekly).toBeDefined(); + expect(result.gemini_weekly.remainingPercentage).toBe(75); + expect(result.claude_gpt_weekly.remainingPercentage).toBe(50); + }); + + it("skips non-weekly buckets", () => { + const data = { + groups: [{ + displayName: "Gemini Models", + buckets: [ + { + bucketId: "gemini-daily-bucket", + displayName: "Daily Limit", + remainingFraction: 0.9, + resetTime: "2026-09-09T00:00:00Z", + }, + ], + }], + }; + const result = parseWeeklyQuotaSummary(data); + expect(result).toEqual({}); + }); + + it("skips disabled weekly buckets", () => { + const data = { + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "gemini-weekly-bucket", + displayName: "Weekly Limit", + remainingFraction: 0.75, + resetTime: "2026-09-15T00:00:00Z", + disabled: true, + }], + }], + }; + const result = parseWeeklyQuotaSummary(data); + expect(result).toEqual({}); + }); + + it("returns empty object for null/undefined input", () => { + expect(parseWeeklyQuotaSummary(null)).toEqual({}); + expect(parseWeeklyQuotaSummary(undefined)).toEqual({}); + expect(parseWeeklyQuotaSummary("string")).toEqual({}); + }); + + it("returns empty object for response with no groups", () => { + expect(parseWeeklyQuotaSummary({})).toEqual({}); + expect(parseWeeklyQuotaSummary({ groups: "not-array" })).toEqual({}); + expect(parseWeeklyQuotaSummary({ quotaSummary: {} })).toEqual({}); + }); + + it("handles groups with no buckets gracefully", () => { + const data = { + groups: [{ displayName: "Gemini Models" }], + }; + expect(parseWeeklyQuotaSummary(data)).toEqual({}); + }); + + it("handles bucket with non-finite remainingFraction", () => { + const data = { + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "weekly-bucket", + displayName: "Weekly", + remainingFraction: "not-a-number", + }], + }], + }; + expect(parseWeeklyQuotaSummary(data)).toEqual({}); + }); + + it("ignores groups that don't match known families", () => { + const data = { + groups: [{ + displayName: "Unknown AI Provider", + buckets: [{ + bucketId: "weekly-bucket", + displayName: "Weekly", + remainingFraction: 0.5, + }], + }], + }; + expect(parseWeeklyQuotaSummary(data)).toEqual({}); + }); +}); + +// — fetchAntigravityWeeklyQuota ——————————————————————————— +describe("fetchAntigravityWeeklyQuota", () => { + beforeEach(() => { + proxyAwareFetch.mockReset(); + _clearWeeklyCache(); + }); + + it("fetches and returns parsed weekly quota on success", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => FULL_RESPONSE, + }); + + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result.gemini_weekly).toBeDefined(); + expect(result.claude_gpt_weekly).toBeDefined(); + }); + + it("returns {} on HTTP 401", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 401 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on HTTP 403", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 403 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on HTTP 404", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 404 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on HTTP 429", async () => { + proxyAwareFetch.mockResolvedValue({ ok: false, status: 429 }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on network error", async () => { + proxyAwareFetch.mockRejectedValue(new Error("network timeout")); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("returns {} on malformed JSON response", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => { throw new SyntaxError("Unexpected token"); }, + }); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + expect(result).toEqual({}); + }); + + it("deduplicates concurrent requests for the same account", async () => { + let resolveResponse; + proxyAwareFetch.mockReturnValue(new Promise(resolve => { + resolveResponse = resolve; + })); + + const p1 = fetchAntigravityWeeklyQuota("token", "project-1"); + const p2 = fetchAntigravityWeeklyQuota("token", "project-1"); + + resolveResponse({ ok: true, json: async () => FULL_RESPONSE }); + + const [r1, r2] = await Promise.all([p1, p2]); + expect(r1).toEqual(r2); + expect(proxyAwareFetch).toHaveBeenCalledTimes(1); + }); + + it("serves cached result within TTL", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => FULL_RESPONSE, + }); + + await fetchAntigravityWeeklyQuota("token", "project-1"); + const result = await fetchAntigravityWeeklyQuota("token", "project-1"); + + expect(proxyAwareFetch).toHaveBeenCalledTimes(1); + expect(result.gemini_weekly).toBeDefined(); + }); + + it("sends correct headers and body", async () => { + proxyAwareFetch.mockResolvedValue({ + ok: true, + json: async () => ({ groups: [] }), + }); + + await fetchAntigravityWeeklyQuota("token", "project-1"); + + expect(proxyAwareFetch).toHaveBeenCalledWith( + "https://daily-cloudcode-pa.googleapis.com/v1internal:retrieveUserQuotaSummary", + expect.objectContaining({ + method: "POST", + headers: expect.objectContaining({ + "Authorization": "Bearer token", + "User-Agent": "antigravity/ide/2.11.0 darwin/arm64", + "Content-Type": "application/json", + "X-Client-Name": "antigravity", + }), + body: JSON.stringify({ project: "project-1" }), + }), + expect.any(Object), + ); + }); +}); + +// — Integration: weekly failure does not affect existing quotas ————— +describe("weekly quota isolation from existing quota", () => { + beforeEach(() => { + proxyAwareFetch.mockReset(); + _clearWeeklyCache(); + }); + + it("existing getAntigravityUsage succeeds even when weekly RPC fails", async () => { + proxyAwareFetch.mockImplementation(async (url) => { + if (url.includes(":loadCodeAssist")) { + return { + ok: true, + status: 200, + json: async () => ({ cloudaicompanionProject: "p1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } }), + }; + } + if (url.includes(":fetchAvailableModels")) { + return { + ok: true, + status: 200, + json: async () => ({ + models: { + "gemini-3.8-flash-high": { + displayName: "Gemini 3.8 Flash (High)", + quotaInfo: { remainingFraction: 0.85, resetTime: "2026-09-15T00:00:00Z" }, + }, + }, + }), + }; + } + if (url.includes(":retrieveUserQuotaSummary")) { + throw new Error("weekly endpoint unavailable"); + } + return { ok: false, status: 404 }; + }); + + const { getAntigravityUsage } = await import("../../open-sse/services/usage/google.js"); + const result = await getAntigravityUsage("token", {}); + + expect(result.quotas["gemini-3.8-flash-high"]).toMatchObject({ + used: 150, + total: 1000, + remainingPercentage: 85, + }); + expect(result.quotas.gemini_weekly).toBeUndefined(); + expect(result.quotas.claude_gpt_weekly).toBeUndefined(); + expect(result.message).toBeUndefined(); + }); + + it("free-tier accounts only show weekly quotas, not per-model short-window quotas", async () => { + proxyAwareFetch.mockImplementation(async (url) => { + if (url.includes(":loadCodeAssist")) { + return { + ok: true, + status: 200, + json: async () => ({ cloudaicompanionProject: "p1", currentTier: { name: "Starter" }, paidTier: { id: "free-tier", name: "Antigravity Starter Quota" } }), + }; + } + if (url.includes(":fetchAvailableModels")) { + return { + ok: true, + status: 200, + json: async () => ({ + models: { + "gemini-3.8-flash-high": { + displayName: "Gemini 3.8 Flash (High)", + quotaInfo: { remainingFraction: 1, resetTime: "2026-09-15T00:00:00Z" }, + }, + "claude-sonnet-4-6": { + displayName: "Claude Sonnet 4.6", + // Missing remainingFraction — free tier exhausted + quotaInfo: { resetTime: "2026-09-13T12:00:00Z" }, + }, + }, + }), + }; + } + if (url.includes(":retrieveUserQuotaSummary")) { + return { + ok: true, + status: 200, + json: async () => ({ + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "gemini-weekly", + displayName: "Weekly Limit Remaining", + remainingFraction: 1, + resetTime: "2026-09-15T00:00:00Z", + }], + }, { + displayName: "Claude and GPT models", + buckets: [{ + bucketId: "3p-weekly", + displayName: "Weekly Limit Remaining", + remainingFraction: 0, + resetTime: "2026-09-13T12:00:00Z", + }], + }], + }), + }; + } + return { ok: false, status: 404 }; + }); + + const { getAntigravityUsage } = await import("../../open-sse/services/usage/google.js"); + const result = await getAntigravityUsage("token", {}); + + // Per-model quotas should be absent (free-tier accounts skip model parsing) + expect(result.quotas["gemini-3.8-flash-high"]).toBeUndefined(); + expect(result.quotas["claude-sonnet-4-6"]).toBeUndefined(); + + // Only weekly quotas should appear + expect(result.quotas.gemini_weekly).toMatchObject({ + used: 0, + total: 1000, + remainingPercentage: 100, + }); + expect(result.quotas.claude_gpt_weekly).toMatchObject({ + used: 1000, + total: 1000, + remainingPercentage: 0, + }); + }); + + it("reconciles weekly quota to 0% when all paid-tier family models are exhausted", async () => { + proxyAwareFetch.mockImplementation(async (url) => { + if (url.includes(":loadCodeAssist")) { + return { + ok: true, + status: 200, + json: async () => ({ cloudaicompanionProject: "p1", currentTier: { name: "Pro" }, paidTier: { id: "g1-pro-tier", name: "Google AI Pro" } }), + }; + } + if (url.includes(":fetchAvailableModels")) { + return { + ok: true, + status: 200, + json: async () => ({ + models: { + "gemini-3.8-flash-high": { + displayName: "Gemini 3.8 Flash (High)", + // Exhausted model: no remainingFraction, future resetTime + quotaInfo: { resetTime: "2026-09-13T12:00:00Z" }, + }, + }, + }), + }; + } + if (url.includes(":retrieveUserQuotaSummary")) { + return { + ok: true, + status: 200, + json: async () => ({ + groups: [{ + displayName: "Gemini Models", + buckets: [{ + bucketId: "gemini-weekly", + displayName: "Weekly Limit Remaining", + remainingFraction: 1, + resetTime: "2026-09-15T00:00:00Z", + }], + }], + }), + }; + } + return { ok: false, status: 404 }; + }); + + const { getAntigravityUsage } = await import("../../open-sse/services/usage/google.js"); + const result = await getAntigravityUsage("token", {}); + + // Per-model quota should show exhausted + expect(result.quotas["gemini-3.8-flash-high"].remainingPercentage).toBe(0); + // Weekly quota should be reconciled to 0% with the family reset time + expect(result.quotas.gemini_weekly).toMatchObject({ + used: 1000, + total: 1000, + remainingPercentage: 0, + resetAt: "2026-09-13T12:00:00.000Z", + }); + }); +}); diff --git a/tests/unit/api-airforce-free-models.test.js b/tests/unit/api-airforce-free-models.test.js new file mode 100644 index 00000000..0b435981 --- /dev/null +++ b/tests/unit/api-airforce-free-models.test.js @@ -0,0 +1,36 @@ +import { describe, it, expect } from "vitest"; +import airforce from "../../open-sse/providers/registry/api-airforce.js"; +import { PROVIDERS, PROVIDER_MODELS } from "../../open-sse/providers/index.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; + +describe("api-airforce free models", () => { + const ids = airforce.models.map((m) => m.id); + + it("registers the three live free models", () => { + expect(ids).toContain("gpt-oss-120b"); + expect(ids).toContain("gpt-oss-20b"); + expect(ids).toContain("kimi-k2.7-code"); + }); + + it("drops the dead catalog ids", () => { + expect(ids).not.toContain("anthropic/claude-3.7-sonnet"); + expect(ids).not.toContain("moonshot/kimi-k2.6"); + expect(ids).not.toContain("google/gemini-2.5-flash"); + }); + + it("is passthrough so any live id resolves", () => { + expect(airforce.passthroughModels).toBe(true); + expect(PROVIDERS["api-airforce"].forceStream).toBe(true); + }); + + it("PROVIDER_MODELS['af'] exposes the new ids", () => { + expect(PROVIDER_MODELS.af.map((m) => m.id)).toEqual(expect.arrayContaining([ + "gpt-oss-120b", "gpt-oss-20b", "kimi-k2.7-code", + ])); + }); + + it("caps resolve for the free ids", () => { + expect(getCapabilitiesForModel("api-airforce", "kimi-k2.7-code").reasoning).toBe(true); + expect(getCapabilitiesForModel("api-airforce", "gpt-oss-120b").reasoning).toBe(true); + }); +}); diff --git a/tests/unit/capabilities.test.js b/tests/unit/capabilities.test.js index ac544d47..84a8643c 100644 --- a/tests/unit/capabilities.test.js +++ b/tests/unit/capabilities.test.js @@ -2,6 +2,24 @@ import { describe, expect, it } from "vitest"; import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; describe("getCapabilitiesForModel", () => { + + it("reports DeepSeek V4.1-Flash ids as vision-capable without dropping their thinking/context", () => { + const v41 = { vision: true, reasoning: true, thinkingFormat: "deepseek", contextWindow: 1000000, maxOutput: 384000 }; + expect(getCapabilitiesForModel(undefined, "deepseek-v4.1-flash")).toMatchObject(v41); + expect(getCapabilitiesForModel("opencode-go", "deepseek-v4.1-flash")).toMatchObject(v41); + expect(getCapabilitiesForModel("openrouter", "deepseek/deepseek-v4.1-flash")).toMatchObject(v41); + // "deepseek-flash" is the GA id for V4.1-Flash on the DeepSeek API; the pattern it + // used to fall through to gives it 128K/64K, which the exact entry keeps. + expect(getCapabilitiesForModel("opencode-go", "deepseek-flash")).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "deepseek", + contextWindow: 128000, + maxOutput: 64000, + }); + // the superseded text-only Flash id stays text-only + expect(getCapabilitiesForModel("opencode-go", "deepseek-v4-flash").vision).toBe(false); + }); const claudeSonnet5Expected = { contextWindow: 1000000, maxOutput: 128000, @@ -62,4 +80,37 @@ describe("getCapabilitiesForModel", () => { expect(getCapabilitiesForModel("kiro", "gpt-5.6-luna-agentic")).toMatchObject(kiroGpt56Expected); expect(getCapabilitiesForModel("kiro", "gpt-5.6-sol-thinking-agentic")).toMatchObject(kiroGpt56Expected); }); + + it("reports Codex GPT 6.0 Astra as a vision and thinking capable model", () => { + expect(getCapabilitiesForModel("codex", "gpt-6-astra")).toMatchObject({ + vision: true, + reasoning: true, + search: true, + thinkingFormat: "openai", + contextWindow: 272000, + maxOutput: 128000, + }); + }); + + it("CommandCode v4.1-flash is vision + effort capable", () => { + expect(getCapabilitiesForModel("commandcode", "deepseek/deepseek-v4.1-flash")).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "commandcode", + thinkingEffortSupported: true, + }); + }); + + it("CommandCode MiniMax-M3 is vision capable", () => { + expect(getCapabilitiesForModel("commandcode", "MiniMaxAI/MiniMax-M3").vision).toBe(true); + }); + + it("CommandCode text-only DeepSeek V4 Flash stays non-vision", () => { + expect(getCapabilitiesForModel("commandcode", "deepseek/deepseek-v4-flash").vision).toBe(false); + expect(getCapabilitiesForModel("commandcode", "deepseek/deepseek-v4-flash")).toMatchObject({ + reasoning: true, + thinkingFormat: "commandcode", + thinkingEffortSupported: true, + }); + }); }); diff --git a/tests/unit/claude-cache-budget-single-object.test.js b/tests/unit/claude-cache-budget-single-object.test.js new file mode 100644 index 00000000..e2680723 --- /dev/null +++ b/tests/unit/claude-cache-budget-single-object.test.js @@ -0,0 +1,222 @@ +// #3795 — a proxy must not dispatch more cache_control blocks than Anthropic +// accepts, and must not lose turns that use single-object content (#3567 interplay). +import { describe, it, expect } from "vitest"; +import { + anchorClaudeCache, + normalizeClaudePassthrough, + prepareClaudeRequest, +} from "../../open-sse/translator/formats/claude.js"; +import { claudeToOpenAIRequest } from "../../open-sse/translator/request/claude-to-openai.js"; + +const CC = { type: "ephemeral" }; +const text = (t, extra = {}) => ({ type: "text", text: t, ...extra }); +const tool = (name, extra = {}) => ({ name, description: "d", input_schema: {}, ...extra }); + +// counts markers incl. single-object content — mirrors the upstream contract +function countMarkers(body) { + let n = 0; + if (Array.isArray(body.system)) for (const b of body.system) if (b?.cache_control) n++; + if (Array.isArray(body.tools)) for (const t of body.tools) if (t?.cache_control) n++; + if (Array.isArray(body.messages)) for (const m of body.messages) { + if (Array.isArray(m?.content)) { + for (const b of m.content) if (b?.cache_control) n++; + } else if (m?.content && typeof m.content === "object" && m.content.cache_control) n++; + } + return n; +} + +describe("cache marker budget and single-block content", () => { + it("never emits more than four markers when the client already spent its budget", () => { + const out = anchorClaudeCache({ + system: [text("s1"), text("s2", { cache_control: CC })], + tools: [tool("t1"), tool("t2", { cache_control: CC })], + messages: [ + { role: "user", content: text("u1", { cache_control: CC }) }, + { role: "assistant", content: text("a1", { cache_control: CC }) }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBeLessThanOrEqual(4); // base: 5 + }); + + it("normalizes a single-object turn in passthrough and anchors it", () => { + const body = { + messages: [ + { role: "user", content: [text("u1")] }, + { role: "assistant", content: text("a1") }, // single object, no marker + { role: "user", content: [text("q")] }, + ], + }; + normalizeClaudePassthrough(body); + const assistant = body.messages.find(m => m.role === "assistant"); + expect(assistant).toBeDefined(); + expect(Array.isArray(assistant.content)).toBe(true); // base: still bare object + expect(assistant.content).toHaveLength(1); + const out = anchorClaudeCache(body); + expect(countMarkers(out)).toBe(1); + }); + + it("keeps a turn whose content is a single object and strips its marker", () => { + const out = prepareClaudeRequest({ + model: "claude-sonnet-5", max_tokens: 100, + system: [text("s1")], + messages: [ + { role: "user", content: text("u1", { cache_control: CC }) }, + { role: "assistant", content: [text("a1")] }, + { role: "user", content: [text("q")] }, + ], + }, "claude"); + const kept = out.messages.filter(m => JSON.stringify(m.content).includes("u1")); + expect(kept.length).toBe(1); // base: 0 (dropped) + expect(kept[0].content).toHaveLength(1); // normalized to array + expect(kept[0].content[0].cache_control).toBeUndefined(); + }); + + it("drops no conversation turn when content is a single text object", () => { + const out = prepareClaudeRequest({ + model: "claude-sonnet-5", max_tokens: 100, + messages: [ + { role: "user", content: text("u1") }, + { role: "assistant", content: text("a1") }, + { role: "user", content: [text("q")] }, + ], + }, "claude"); + expect(out.messages.length).toBe(3); // base: 1 + expect(Array.isArray(out.messages[0].content)).toBe(true); + }); + + it("re-anchors the last assistant turn even when it uses single-object content", () => { + const out = prepareClaudeRequest({ + model: "claude-sonnet-5", max_tokens: 100, + messages: [ + { role: "user", content: [text("u1")] }, + { role: "assistant", content: text("a1") }, + { role: "user", content: [text("q")] }, + ], + }, "claude"); + expect(countMarkers(out)).toBe(1); // base: 0 + }); + + it("keeps a marked single-object turn when the marker budget is spent", () => { + const body = { + system: [text("s1", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC })], + messages: [ + { role: "user", content: [text("c1"), text("c2")] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: text("u1", { cache_control: CC }) }, + { role: "user", content: [text("q")] }, + ], + }; + normalizeClaudePassthrough(body); + const out = anchorClaudeCache(body); + const kept = out.messages.filter(m => JSON.stringify(m.content).includes("u1")); + expect(kept.length).toBe(1); + expect(Array.isArray(kept[0].content)).toBe(true); // base: bare object survives + expect(kept[0].content).toHaveLength(1); + expect(kept[0].content[0].cache_control).toBeUndefined(); + const ctx = out.messages.find(m => JSON.stringify(m.content).includes("c1")); + expect(ctx.content).toEqual([text("c1"), text("c2")]); + expect(countMarkers(out)).toBeLessThanOrEqual(4); // fixed: 3 + }); + + it("keeps single-object turns on the claude-to-openai leg", () => { + const out = claudeToOpenAIRequest("m", { + messages: [ + { role: "user", content: text("u1") }, + { role: "assistant", content: { type: "image", source: { type: "base64", media_type: "image/png", data: "iVBORw0KGgo=" } } }, + ], + }, false); + expect(out.messages.some(m => JSON.stringify(m.content).includes("u1"))).toBe(true); // base: dropped + const img = out.messages.find(m => m.role === "assistant"); + expect(JSON.stringify(img.content)).toContain("image_url"); // base: dropped + }); + + it("keeps a bare-object user turn folded with a mid-conversation system message", () => { + const body = { + messages: [ + { role: "user", content: text("u1") }, + { role: "system", content: [text("reminder")] }, + { role: "user", content: [text("q")] }, + ], + }; + normalizeClaudePassthrough(body); + const first = body.messages[0]; + expect(Array.isArray(first.content)).toBe(true); + expect(JSON.stringify(first.content).includes("u1")).toBe(true); // pre-hoist: fold zeroes bare-object content + expect(JSON.stringify(first.content).includes("reminder")).toBe(true); + }); + it("prunes a client body that already carries five markers down to four", () => { + const out = anchorClaudeCache({ + system: [text("s1"), text("s2", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC }), tool("t2", { cache_control: CC })], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBe(4); // pre-fix: 5 forwarded unchanged + expect(out.system[0].cache_control).toBeUndefined(); // earliest marker pruned + }); + + // A spent budget must not cost the head anchors their 1h TTL: system/tools are + // the whole point of re-anchoring, and a 5m fallback silently halves the cache + // lifetime on exactly the requests that already cached aggressively. + it("keeps the 1h head anchors when the client spent the whole budget", () => { + const out = anchorClaudeCache({ + system: [text("s1"), text("s2", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC }), tool("t2")], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBeLessThanOrEqual(4); + expect(out.system.at(-1).cache_control?.ttl).toBe("1h"); // pre-fix: fell back to 5m + expect(out.tools.at(-1).cache_control?.ttl).toBe("1h"); // pre-fix: fell back to 5m + }); + + it("keeps the 1h head anchors on an over-budget body", () => { + const out = anchorClaudeCache({ + system: [text("s1", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC }), tool("t2")], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("u2", { cache_control: CC })] }, + { role: "assistant", content: [text("a2", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBe(4); + expect(out.system.at(-1).cache_control?.ttl).toBe("1h"); + expect(out.tools.at(-1).cache_control?.ttl).toBe("1h"); + }); + + it("strips a marker from a deferred tool even when the budget is spent", () => { + const out = anchorClaudeCache({ + system: [text("s1", { cache_control: CC })], + tools: [tool("t1", { cache_control: CC, defer_loading: true })], + messages: [ + { role: "user", content: [text("u1", { cache_control: CC })] }, + { role: "assistant", content: [text("a1", { cache_control: CC })] }, + { role: "user", content: [text("q")] }, + ], + }); + expect(countMarkers(out)).toBeLessThanOrEqual(4); + const deferred = out.tools.find(t => t.defer_loading); + expect(deferred?.cache_control).toBeUndefined(); // pre-fix: invalid marker forwarded + }); + + it("keeps a bare-object system reminder on the claude-to-openai leg", () => { + const out = claudeToOpenAIRequest("m", { + messages: [ + { role: "user", content: "hi" }, + { role: "system", content: text("be brief") }, + ], + }, false); + expect(JSON.stringify(out.messages)).toContain("be brief"); // pre-fix: turn dropped + }); +}); diff --git a/tests/unit/claude-header-forwarding.test.js b/tests/unit/claude-header-forwarding.test.js index 840a8559..d813f356 100644 --- a/tests/unit/claude-header-forwarding.test.js +++ b/tests/unit/claude-header-forwarding.test.js @@ -187,6 +187,46 @@ describe("DefaultExecutor.buildHeaders() — anthropic-compatible stripping", () headers["Anthropic-Version"] || headers["anthropic-version"]; expect(hasVersion).toBeDefined(); }); + + // A node fronting Anthropic (rotating multi-account proxy, corporate gateway) + // needs the same beta flags the `claude` provider sends. Without + // context-management-2025-06-27 upstream answers HTTP 400 + // "context_management: Extra inputs are not permitted" and the combo falls + // through to the next model without anyone noticing. + it("sends context-management beta for a Claude model on a custom host", () => { + const executor = new DefaultExecutor("anthropic-compatible-custom"); + const headers = executor.buildHeaders( + { + apiKey: "key", + providerSpecificData: { baseUrl: "https://myproxy.example.com/v1" }, + }, + true, + undefined, + "claude-opus-5" + ); + + const betaFlags = (headers["Anthropic-Beta"] || headers["anthropic-beta"] || "") + .split(",").map(s => s.trim()); + expect(betaFlags).toContain("context-management-2025-06-27"); + // The first-party identity flag is still stripped for a non-Anthropic host. + expect(betaFlags).not.toContain("claude-code-20250219"); + }); + + it("gates the beta flags on the model id, not the provider prefix", () => { + const executor = new DefaultExecutor("anthropic-compatible-custom"); + const headers = executor.buildHeaders( + { + apiKey: "key", + providerSpecificData: { baseUrl: "https://myproxy.example.com/v1" }, + }, + true, + undefined, + "kimi-k3" + ); + + const betaVal = headers["Anthropic-Beta"] || headers["anthropic-beta"] || ""; + expect(betaVal).not.toContain("context-management-2025-06-27"); + }); }); // ─── proxyFetch anthropicFetch routing ──────────────────────────────────────── diff --git a/tests/unit/cline-auth.test.js b/tests/unit/cline-auth.test.js new file mode 100644 index 00000000..e1eaae20 --- /dev/null +++ b/tests/unit/cline-auth.test.js @@ -0,0 +1,40 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + getClineAccessToken, + getClineAuthorizationHeader, +} from "../../open-sse/shared/clineAuth.js"; + +test("getClineAccessToken keeps an existing workos: prefix", () => { + const token = "workos:eyJhbGciOiJSUzI1NiJ9.eyJwYXAiJ9"; + assert.equal(getClineAccessToken(token), token); + assert.equal(getClineAccessToken(` ${token} `), token); +}); + +test("getClineAccessToken prefixes a bare WorkOS JWT with workos:", () => { + const jwt = "eyJhbGciOiJSUzI1NiJ9.eyJwYXAiJ9"; + assert.equal(getClineAccessToken(jwt), `workos:${jwt}`); +}); + +test("getClineAccessToken does NOT prefix ClinePass API keys", () => { + // ClinePass API keys are opaque strings (e.g. clp_…). Sending them as + // `workos:clp_…` makes api.cline.bot respond 401. + assert.equal(getClineAccessToken("clp_1234567890abcdef"), "clp_1234567890abcdef"); + assert.equal(getClineAccessToken("sk-9r-abcdef"), "sk-9r-abcdef"); + assert.equal(getClineAccessToken(""), ""); + assert.equal(getClineAccessToken(" "), ""); + assert.equal(getClineAccessToken(undefined), ""); + assert.equal(getClineAccessToken(null), ""); +}); + +test("getClineAuthorizationHeader builds a Bearer header without double prefixing", () => { + assert.equal(getClineAuthorizationHeader("clp_abc"), "Bearer clp_abc"); + assert.equal( + getClineAuthorizationHeader("eyJpeg.eyJbG"), + "Bearer workos:eyJpeg.eyJbG" + ); + assert.equal( + getClineAuthorizationHeader("workos:eyJpeg.eyJbG"), + "Bearer workos:eyJpeg.eyJbG" + ); +}); \ No newline at end of file diff --git a/tests/unit/cline-free-models-envelope.test.js b/tests/unit/cline-free-models-envelope.test.js new file mode 100644 index 00000000..8ae0fa89 --- /dev/null +++ b/tests/unit/cline-free-models-envelope.test.js @@ -0,0 +1,297 @@ +// Cline free models (z-ai/glm-5.3-flash, deepseek-v4-flash) wrap non-stream +// chat completions in {"success":true,"data":{...choices...}} on +// https://api.cline.bot/api/v1/chat/completions. Both the UI model-test ping +// (src/app/api/models/test/ping.js) and the proxy non-stream path +// (open-sse/handlers/chatCore/nonStreamingHandler.js) read `choices` at the +// top level, so enveloped choices are invisible ("Provider returned no +// completion choices for this model"). These tests pin the unwrap behavior. + +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; + +// Mock the heavy Next.js-dependent imports BEFORE importing ping.js +// (same pattern as tests/unit/ping-reasoning-models-3010.test.js). +vi.mock("@/lib/localDb", () => ({ getApiKeys: vi.fn(async () => [{ key: "test-key", isActive: true }]) })); +vi.mock("@/shared/constants/config", () => ({ UPDATER_CONFIG: { appPort: 20127 } })); +vi.mock("@/shared/utils/machineId", () => ({ getConsistentMachineId: vi.fn(async () => "cli-token") })); +// requestDetail.js imports from @/lib/usageDb.js too, so one mock covers both +// the handler and its usage/detail helpers. +vi.mock("@/lib/usageDb.js", () => ({ + appendRequestLog: vi.fn(async () => {}), + saveRequestDetail: vi.fn(async () => {}), + saveRequestUsage: vi.fn(async () => {}), +})); + +const { pingModelByKind } = await import("../../src/app/api/models/test/ping.js"); +const { handleNonStreamingResponse } = await import("../../open-sse/handlers/chatCore/nonStreamingHandler.js"); + +// The proxy adds a 2000-token headroom buffer to usage before returning it +// to the client (addBufferToUsage), so response-body usage is input + 2000. +// The usage recorded via saveRequestUsage is the unbuffered extraction — +// asserting on it proves the unwrap ran before usage extraction. +const { saveRequestUsage } = await import("@/lib/usageDb.js"); + +describe("cline free-models {success,data} envelope", () => { + let fetchMock; + + beforeEach(() => { + fetchMock = vi.fn(); + vi.stubGlobal("fetch", fetchMock); + }); + + afterEach(() => { + vi.unstubAllGlobals(); + }); + + function jsonResponse(obj) { + return { + ok: true, + status: 200, + text: async () => JSON.stringify(obj), + json: async () => obj, + }; + } + + it("ping: enveloped success unwraps to ok:true (regression for reported error)", async () => { + fetchMock.mockResolvedValue( + jsonResponse({ success: true, data: { choices: [{ message: { content: "OK" } }] } }) + ); + const result = await pingModelByKind("cl/z-ai/glm-5.3-flash", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(true); + }); + + it("ping: enveloped reasoning-only response still soft-passes with note", async () => { + fetchMock.mockResolvedValue( + jsonResponse({ + success: true, + data: { + choices: [ + { + finish_reason: "length", + message: { content: "", reasoning: "The user said hi" }, + }, + ], + }, + }) + ); + const result = await pingModelByKind("cl/z-ai/glm-5.3-flash", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(true); + expect(result.note).toMatch(/reasoning-only/); + }); + + it("ping: error envelope passes through without unwrap", async () => { + const body = { error: "empty response content", success: false }; + fetchMock.mockResolvedValue({ + ok: false, + status: 500, + text: async () => JSON.stringify(body), + json: async () => body, + }); + const result = await pingModelByKind("cl/z-ai/glm-5.3-flash", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(false); + expect(result.error).toMatch(/empty response content/); + }); + + it("ping: bare (un-enveloped) body still passes", async () => { + fetchMock.mockResolvedValue(jsonResponse({ choices: [{ message: { content: "Hello!" } }] })); + const result = await pingModelByKind("openai/gpt-4o", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(true); + }); + + it("ping: does not unwrap for a provider that did not opt in", async () => { + fetchMock.mockResolvedValue( + jsonResponse({ success: true, data: { choices: [{ message: { content: "OK" } }] } }) + ); + const result = await pingModelByKind("openai/gpt-4o", "llm", "http://127.0.0.1:20127"); + expect(result.ok).toBe(false); + expect(result.error).toMatch(/no completion choices/i); + }); +}); + +describe("cline free-models envelope in nonStreamingHandler", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + function stubLogger() { + return { logProviderResponse() {}, logConvertedResponse() {} }; + } + + function callHandler(providerResponse, provider = "cline") { + return handleNonStreamingResponse({ + providerResponse, + provider, + model: "z-ai/glm-5.3-flash", + sourceFormat: "openai", + targetFormat: "openai", + body: { stream: false }, + stream: false, + translatedBody: null, + finalBody: null, + requestStartTime: Date.now(), + connectionId: "c1", + apiKey: "k", + clientRawRequest: null, + onRequestSuccess: () => {}, + reqLogger: stubLogger(), + toolNameMap: null, + customToolNames: null, + trackDone: () => {}, + appendLog: () => {}, + pxpipe: null, + reqTag: "t", + log: null, + }); + } + + it("unwraps the {success,data} envelope before usage extraction and translation", async () => { + const providerResponse = new Response( + JSON.stringify({ + success: true, + data: { + choices: [{ message: { content: "Hi" } }], + usage: { prompt_tokens: 5, completion_tokens: 2 }, + }, + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + const result = await callHandler(providerResponse); + expect(result.success).toBe(true); + const body = await result.response.json(); + expect(body.choices).toBeDefined(); + expect(body.choices[0].message.content).toBe("Hi"); + expect(body.success).toBeUndefined(); + expect(saveRequestUsage).toHaveBeenCalledTimes(1); + expect(saveRequestUsage.mock.calls[0][0].tokens).toMatchObject({ + prompt_tokens: 5, + completion_tokens: 2, + }); + expect(body.usage.prompt_tokens).toBe(2005); + }); + + it("passes a bare (non-enveloped) body through unchanged", async () => { + const providerResponse = new Response( + JSON.stringify({ + choices: [{ message: { content: "Hi" } }], + usage: { prompt_tokens: 3, completion_tokens: 1 }, + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + const result = await callHandler(providerResponse); + expect(result.success).toBe(true); + const body = await result.response.json(); + expect(body.choices[0].message.content).toBe("Hi"); + expect(saveRequestUsage).toHaveBeenCalledTimes(1); + expect(saveRequestUsage.mock.calls[0][0].tokens).toMatchObject({ + prompt_tokens: 3, + completion_tokens: 1, + }); + expect(body.usage.prompt_tokens).toBe(2003); + }); + + // The unwrap is opt-in via transport.quirks.clineEnvelope so it can never + // rewrite another provider's body — including one that happens to return + // {"success":true,"data":...} for its own reasons. + it("leaves an enveloped body untouched for a provider that did not opt in", async () => { + const providerResponse = new Response( + JSON.stringify({ + success: true, + data: { choices: [{ message: { content: "Hi" } }] }, + }), + { status: 200, headers: { "Content-Type": "application/json" } } + ); + const result = await callHandler(providerResponse, "openai"); + const body = await result.response.json(); + expect(body.success).toBe(true); + expect(body.data.choices[0].message.content).toBe("Hi"); + expect(body.choices).toBeUndefined(); + }); +}); + +describe("cline /api/v1/models aggregation (resolveClineModels vs resolveClinepassModels)", () => { + const API_MODELS_URL = "https://api.cline.bot/api/v1/models"; + + const API_RESPONSE = [ + { id: "cline-pass/deepseek-v4-flash", name: "DeepSeek V4 Flash" }, + { id: "cline-pass/glm-5.2", name: "GLM-5.2" }, + { id: "z-ai/glm-5.3-flash", name: "GLM-5.3 Flash" }, + { id: "z-ai/deepseek-v4-flash", name: "DeepSeek V4 Flash (Free)" }, + ]; + + let fetchMock; + + beforeEach(() => { + fetchMock = vi.fn(); + vi.stubGlobal("fetch", fetchMock); + }); + + afterEach(() => { + vi.unstubAllGlobals(); + }); + + it("resolveClineModels returns all models (including free-tier)", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => API_RESPONSE, + }); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result).not.toBeNull(); + expect(result.models).toHaveLength(4); + const ids = result.models.map((m) => m.id); + expect(ids).toContain("cline-pass/deepseek-v4-flash"); + expect(ids).toContain("z-ai/glm-5.3-flash"); + expect(ids).toContain("z-ai/deepseek-v4-flash"); + }); + + it("resolveClinepassModels returns only cline-pass/ models", async () => { + const { resolveClinepassModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => API_RESPONSE, + }); + const result = await resolveClinepassModels({ accessToken: "test-token" }); + expect(result).not.toBeNull(); + expect(result.models).toHaveLength(2); + const ids = result.models.map((m) => m.id); + expect(ids).toContain("cline-pass/deepseek-v4-flash"); + expect(ids).toContain("cline-pass/glm-5.2"); + expect(ids).not.toContain("z-ai/glm-5.3-flash"); + }); + + it("resolveClineModels unwraps {success,data} envelope", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => ({ success: true, data: API_RESPONSE }), + }); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result).not.toBeNull(); + expect(result.models).toHaveLength(4); + }); + + it("resolveClineModels returns null when no token", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + const result = await resolveClineModels({}); + expect(result).toBeNull(); + }); + + it("resolveClineModels returns null on fetch error", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockRejectedValue(new Error("network error")); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result).toBeNull(); + }); + + it("resolveClineModels returns {id,name} shape", async () => { + const { resolveClineModels } = await import("../../open-sse/services/clinepassModels.js"); + fetchMock.mockResolvedValue({ + ok: true, + json: async () => API_RESPONSE, + }); + const result = await resolveClineModels({ accessToken: "test-token" }); + expect(result.models[0]).toHaveProperty("id"); + expect(result.models[0]).toHaveProperty("name"); + expect(typeof result.models[0].id).toBe("string"); + expect(typeof result.models[0].name).toBe("string"); + }); +}); diff --git a/tests/unit/codex-auto-review-routing.test.js b/tests/unit/codex-auto-review-routing.test.js new file mode 100644 index 00000000..e96b0fef --- /dev/null +++ b/tests/unit/codex-auto-review-routing.test.js @@ -0,0 +1,44 @@ +import { describe, expect, it } from "vitest"; + +import { + getDefaultModel, + getModelQuotaFamily, + getModelUpstreamId, + getProviderModels, +} from "../../open-sse/config/providerModels.js"; +import { getModelInfoCore } from "../../open-sse/services/model.js"; + +// Codex CLI's auto-review sends the bare model id "codex-auto-review". Before #1398 it fell +// through prefix inference to the "openai" default and failed with +// "No active credentials for provider: openai". +describe("codex auto-review routing (#1398)", () => { + it("routes the bare Codex auto-review model to the OAuth Codex provider", async () => { + await expect(getModelInfoCore("codex-auto-review", {})).resolves.toEqual({ + provider: "codex", + model: "codex-auto-review", + }); + }); + + it("exposes Codex auto-review as a review-quota Codex model", () => { + const autoReview = getProviderModels("cx").find( + (model) => model.id === "codex-auto-review", + ); + + expect(autoReview).toBeTruthy(); + expect(autoReview.name).toBe("Codex Auto Review"); + expect(getModelQuotaFamily("cx", "codex-auto-review")).toBe("review"); + }); + + // getModelUpstreamId strips CODEX_REVIEW_SUFFIX from unregistered "cx" ids, which would send + // "codex-auto" upstream. This model is not a derived review variant, so it must go out verbatim. + it("forwards the id upstream without stripping the -review suffix", () => { + expect(getModelUpstreamId("cx", "codex-auto-review")).toBe( + "codex-auto-review", + ); + }); + + // Registering it must not push it to the front of the cx list — getDefaultModel takes models[0]. + it("does not become the default Codex model", () => { + expect(getDefaultModel("cx")).not.toBe("codex-auto-review"); + }); +}); diff --git a/tests/unit/codex-image-models.test.js b/tests/unit/codex-image-models.test.js new file mode 100644 index 00000000..9a4e9c44 --- /dev/null +++ b/tests/unit/codex-image-models.test.js @@ -0,0 +1,76 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { getModelsByProviderId, getModelType, isValidModel } from "../../open-sse/config/providerModels.js"; +import { getModelInfoCore } from "../../open-sse/services/model.js"; +import { handleImageGenerationCore } from "../../open-sse/handlers/imageGenerationCore.js"; + +const models = ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra"]; + +afterEach(() => vi.unstubAllGlobals()); + +describe("Codex GPT-5.6 image models", () => { + it.each(models)("exposes %s-image as an image model while retaining its chat entry", (model) => { + const catalog = getModelsByProviderId("codex"); + expect(catalog.filter((entry) => entry.id === `${model}-image`)).toHaveLength(1); + expect(catalog.find((entry) => entry.id === `${model}-image`)).toMatchObject({ + kind: "image", + capabilities: ["text2img", "edit"], + params: ["size", "quality", "background", "image_detail", "output_format"], + }); + expect(isValidModel("cx", `${model}-image`)).toBe(true); + expect(getModelType("cx", `${model}-image`)).toBe("image"); + expect(catalog.find((entry) => entry.id === model)).toBeDefined(); + expect(getModelType("cx", model)).not.toBe("image"); + }); + + it.each(models)("routes %s-image edits and streams image events", async (model) => { + const events = [ + ["response.image_generation_call.partial_image", { partial_image_b64: "cGFydGlhbA==", partial_image_index: 0 }], + ["response.output_item.done", { item: { type: "image_generation_call", result: "ZmluYWw=" } }], + ]; + const fetchMock = vi.fn().mockResolvedValue(new Response( + events.map(([event, data]) => `event: ${event}\ndata: ${JSON.stringify(data)}\n\n`).join(""), + { headers: { "Content-Type": "text/event-stream" } }, + )); + vi.stubGlobal("fetch", fetchMock); + const onRequestSuccess = vi.fn(); + const modelInfo = await getModelInfoCore(`cx/${model}-image`); + expect(modelInfo).toEqual({ provider: "codex", model: `${model}-image` }); + expect(await getModelInfoCore(`codex/${model}-image`)).toEqual(modelInfo); + + const result = await handleImageGenerationCore({ + modelInfo, + body: { + prompt: "Make the square blue", + image: "data:image/png;base64,cmVmZXJlbmNl", + image_detail: "low", + size: "1024x1024", + quality: "high", + background: "transparent", + output_format: "WEBP", + }, + credentials: { accessToken: "test-token" }, + streamToClient: true, + onRequestSuccess, + }); + + expect(result.success).toBe(true); + expect(result.response.status).toBe(200); + expect(result.response.headers.get("content-type")).toBe("text/event-stream"); + const [url, options] = fetchMock.mock.calls[0]; + expect(url).toBe("https://chatgpt.com/backend-api/codex/responses"); + const upstreamBody = JSON.parse(options.body); + expect(upstreamBody.model).toBe(model); + expect(upstreamBody.tools).toEqual([{ + type: "image_generation", output_format: "webp", size: "1024x1024", + quality: "high", background: "transparent", + }]); + expect(upstreamBody.input[0].content).toContainEqual({ + type: "input_image", image_url: "data:image/png;base64,cmVmZXJlbmNl", detail: "low", + }); + const stream = await result.response.text(); + expect(stream).toContain('event: partial_image\ndata: {"b64_json":"cGFydGlhbA==","index":0}'); + expect(stream).toContain("event: done\n"); + expect(stream).toContain('"data":[{"b64_json":"ZmluYWw="}]'); + expect(onRequestSuccess).toHaveBeenCalledTimes(1); + }); +}); diff --git a/tests/unit/codex-reset-credits.test.js b/tests/unit/codex-reset-credits.test.js index c8b4c6fd..43846248 100644 --- a/tests/unit/codex-reset-credits.test.js +++ b/tests/unit/codex-reset-credits.test.js @@ -91,6 +91,17 @@ describe("Codex reset credits", () => { }); }); + it("surfaces structured upstream errors as readable messages", async () => { + mocks.proxyAwareFetch.mockResolvedValue({ + ok: false, + status: 403, + json: async () => ({ error: { message: "Reset credits are unavailable for this account" } }), + }); + + const { getCodexRateLimitResetCredits } = await import("../../open-sse/services/usage/codex.js"); + await expect(getCodexRateLimitResetCredits("token")).rejects.toThrow("Reset credits are unavailable for this account"); + }); + it("GET refreshes OAuth credentials before returning reset credit details", async () => { const connection = { id: "conn_1", diff --git a/tests/unit/codex-tool-normalization.test.js b/tests/unit/codex-tool-normalization.test.js index d1b5901e..c7193898 100644 --- a/tests/unit/codex-tool-normalization.test.js +++ b/tests/unit/codex-tool-normalization.test.js @@ -118,6 +118,69 @@ describe("CodexExecutor tool normalization", () => { ]); }); + it("strips only Unicode-property patterns rejected by Codex", () => { + const unicodePattern = "^(?!__.*__$)[^\\p{Cc}\\p{Cf}\\p{Zl}\\p{Zp}]{1,200}$"; + const validPattern = "^[a-z][a-z0-9_-]{0,31}$"; + const sourceParameters = { + type: "object", + properties: { + artifact: { + type: "object", + properties: { + name: { type: "string", pattern: unicodePattern }, + slug: { type: "string", pattern: validPattern }, + }, + }, + // A property named "pattern" is data, not the schema keyword. + pattern: { type: "string", pattern: validPattern }, + }, + allOf: [{ properties: { title: { type: "string", pattern: unicodePattern } } }], + }; + const tools = normalizeTools([{ + type: "function", + name: "Artifact", + parameters: sourceParameters, + }]); + + expect(tools[0].parameters.properties.artifact.properties.name.pattern).toBeUndefined(); + expect(tools[0].parameters.properties.artifact.properties.slug.pattern).toBe(validPattern); + expect(tools[0].parameters.properties.pattern.pattern).toBe(validPattern); + expect(tools[0].parameters.allOf[0].properties.title.pattern).toBeUndefined(); + // Copy-on-write: the caller's schema remains available for another provider. + expect(sourceParameters.properties.artifact.properties.name.pattern).toBe(unicodePattern); + }); + + it("keeps escaped literal property text and schema identity when no strip is needed", () => { + const parameters = { + type: "object", + properties: { + literal: { type: "string", pattern: "^\\\\p{Cc}$" }, + simple: { type: "string", pattern: "^[A-Z]+$" }, + }, + }; + const tools = normalizeTools([{ type: "function", name: "probe", parameters }]); + + expect(tools[0].parameters).toBe(parameters); + expect(tools[0].parameters.properties.literal.pattern).toBe("^\\\\p{Cc}$"); + }); + + it("sanitizes nested namespace function schemas", () => { + const tools = normalizeTools([{ + type: "namespace", + name: "agent", + tools: [{ + type: "function", + name: "Artifact", + parameters: { + type: "object", + properties: { name: { type: "string", pattern: "^\\p{Cc}+$" } }, + }, + }], + }]); + + expect(tools[0].tools[0].parameters.properties.name.pattern).toBeUndefined(); + }); + it("preserves custom freeform tools with format payloads", () => { const tools = normalizeTools([ { diff --git a/tests/unit/commandcode-executor.test.js b/tests/unit/commandcode-executor.test.js index bd0a23cd..f498b39c 100644 --- a/tests/unit/commandcode-executor.test.js +++ b/tests/unit/commandcode-executor.test.js @@ -132,6 +132,46 @@ describe("inspectAndWrapCommandCodeResponse", () => { expect(text).toContain("Hello from Laguna"); expect(text).toContain("data: [DONE]"); }); + + it("retries when initial stream yields an error and succeeds on second attempt", async () => { + let callCount = 0; + const executor = new CommandCodeExecutor(); + + // Override execute on instance to test retry behavior + executor.execute = async (opts) => { + const maxRetries = 2; + for (let attempt = 0; attempt <= maxRetries; attempt++) { + callCount++; + let rawResponse; + if (callCount === 1) { + rawResponse = new Response(createNdjsonStream([ + JSON.stringify({ + type: "error", + error: { type: "server_error", message: "Network connection lost." } + }) + "\n" + ]), { status: 200, headers: { "Content-Type": "text/event-stream" } }); + } else { + rawResponse = new Response(createNdjsonStream([ + JSON.stringify({ type: "start" }) + "\n", + JSON.stringify({ type: "text-delta", text: "Recovered from lost connection" }) + "\n", + JSON.stringify({ type: "finish" }) + "\n" + ]), { status: 200, headers: { "Content-Type": "text/event-stream" } }); + } + + const wrappedResponse = await inspectAndWrapCommandCodeResponse(rawResponse, opts.model); + if (!wrappedResponse.ok && attempt < maxRetries) { + continue; + } + return { response: wrappedResponse }; + } + }; + + const res = await executor.execute({ model: "deepseek/deepseek-v4.1-flash" }); + expect(res.response.ok).toBe(true); + expect(callCount).toBe(2); + const text = await res.response.text(); + expect(text).toContain("Recovered from lost connection"); + }); }); describe("CommandCode in Combo Fallback", () => { diff --git a/tests/unit/commandcode-usage.test.js b/tests/unit/commandcode-usage.test.js new file mode 100644 index 00000000..5f38dc47 --- /dev/null +++ b/tests/unit/commandcode-usage.test.js @@ -0,0 +1,135 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(), +})); + +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; +import { getUsageForProvider } from "../../open-sse/services/usage.js"; +import { + USAGE_SUPPORTED_PROVIDERS, + USAGE_APIKEY_PROVIDERS, +} from "../../src/shared/constants/providers.js"; +import { parseQuotaData } from "../../src/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js"; + +const BASE = "https://api.commandcode.ai"; + +function jsonResponse(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +const WHOAMI = { + user: { name: "Hieu", email: "hieu@example.com" }, + org: { id: "org_1", name: "personal" }, +}; +const CREDITS = { + credits: { monthlyCredits: 12.5, purchasedCredits: 1, freeCredits: 0.5 }, + windowLimits: { + fiveHour: { used: 2, cap: 10, resetAt: Date.now() + 3_600_000, exceeded: false }, + weekly: { used: 20, cap: 70, resetAt: Date.now() + 86_400_000, exceeded: false }, + }, +}; +const SUBS = { + data: { + planId: "individual-goat", + currentPeriodStart: "2026-09-01T00:00:00.000Z", + currentPeriodEnd: "2026-10-01T00:00:00.000Z", + }, +}; + +function mockHappyPath() { + proxyAwareFetch.mockImplementation(async (url) => { + const u = String(url); + if (u.includes("/alpha/whoami")) return jsonResponse(WHOAMI); + if (u.includes("/alpha/billing/credits")) return jsonResponse(CREDITS); + if (u.includes("/alpha/billing/subscriptions")) return jsonResponse(SUBS); + return jsonResponse({ error: "unexpected " + u }, 404); + }); +} + +describe("commandcode registry usage flags", () => { + it("is listed for apikey quota dashboard", () => { + expect(USAGE_SUPPORTED_PROVIDERS).toContain("commandcode"); + expect(USAGE_APIKEY_PROVIDERS).toContain("commandcode"); + }); +}); + +describe("getUsageForProvider(commandcode)", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + it("returns a message when apiKey is missing", async () => { + const usage = await getUsageForProvider({ provider: "commandcode" }); + expect(usage.message).toMatch(/api key/i); + expect(proxyAwareFetch).not.toHaveBeenCalled(); + }); + + it("GETs whoami, credits, and subscriptions with Bearer apiKey", async () => { + mockHappyPath(); + const usage = await getUsageForProvider({ + provider: "commandcode", + apiKey: "user_test", + }); + + expect(usage.message).toBeUndefined(); + expect(usage.plan).toBe("GOAT"); + const urls = proxyAwareFetch.mock.calls.map(([url]) => String(url)); + expect(urls.some((u) => u.startsWith(`${BASE}/alpha/whoami`))).toBe(true); + expect(urls.some((u) => u.includes("/alpha/billing/credits") && u.includes("orgId=org_1"))).toBe(true); + expect(urls.some((u) => u.includes("/alpha/billing/subscriptions") && u.includes("orgId=org_1"))).toBe(true); + expect(proxyAwareFetch.mock.calls[0][1].headers.Authorization).toBe("Bearer user_test"); + }); + + it("maps remaining credits vs plan cap and rate windows", async () => { + mockHappyPath(); + const usage = await getUsageForProvider({ + provider: "commandcode", + apiKey: "user_test", + }); + + // remaining = 12.5 + 1 + 0.5 = 14; cap GOAT = 70; used = 56 + expect(usage.quotas.Credits).toMatchObject({ + used: 56, + total: 70, + unlimited: false, + }); + expect(usage.quotas["Session (5h)"]).toMatchObject({ + used: 2, + total: 10, + unlimited: false, + }); + expect(usage.quotas.Weekly).toMatchObject({ + used: 20, + total: 70, + }); + expect(new Date(usage.quotas.Credits.resetAt).toISOString()).toBe("2026-10-01T00:00:00.000Z"); + }); + + it("returns an auth message on 401", async () => { + proxyAwareFetch.mockResolvedValueOnce(jsonResponse({ error: "unauthorized" }, 401)); + const usage = await getUsageForProvider({ + provider: "commandcode", + apiKey: "bad", + }); + expect(usage.message).toMatch(/auth|key|login/i); + }); +}); + +describe("parseQuotaData(commandcode)", () => { + it("forwards used/total/resetAt for the dashboard table", () => { + const rows = parseQuotaData("commandcode", { + plan: "GOAT", + quotas: { + Credits: { used: 56, total: 70, resetAt: "2026-10-01T00:00:00.000Z" }, + "Session (5h)": { used: 2, total: 10, resetAt: "2026-09-16T10:00:00.000Z" }, + }, + }); + expect(rows).toHaveLength(2); + expect(rows[0]).toMatchObject({ name: "Credits", used: 56, total: 70 }); + expect(rows[1]).toMatchObject({ name: "Session (5h)", used: 2, total: 10 }); + }); +}); diff --git a/tests/unit/cowork-mcp-ssrf-guard.test.js b/tests/unit/cowork-mcp-ssrf-guard.test.js new file mode 100644 index 00000000..663c13b8 --- /dev/null +++ b/tests/unit/cowork-mcp-ssrf-guard.test.js @@ -0,0 +1,61 @@ +/** + * SSRF guard on POST /api/cli-tools/cowork-mcp-tools (#3782). + * + * Remote callers must not be able to force server-side fetches to + * internal URLs; local-host use (self-hosted MCP servers) keeps working. + */ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +vi.mock("next/server", () => ({ + NextResponse: { + json: (body, init) => + new Response(JSON.stringify(body), { + status: init?.status ?? 200, + headers: { "content-type": "application/json" }, + }), + }, +})); + +const { POST } = await import( + "../../src/app/api/cli-tools/cowork-mcp-tools/route.js" +); + +function remoteRequest(url) { + return new Request("http://gateway.example.com/api/cli-tools/cowork-mcp-tools", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ url }), + }); +} + +describe("cowork-mcp-tools SSRF guard", () => { + beforeEach(() => { + vi.restoreAllMocks(); + }); + + it("rejects loopback URLs from remote callers without fetching", async () => { + const fetchSpy = vi.spyOn(globalThis, "fetch"); + const res = await POST(remoteRequest("http://127.0.0.1:18731/internal-admin")); + expect(res.status).toBe(400); + expect(await res.json()).toEqual({ error: "URL not allowed" }); + expect(fetchSpy).not.toHaveBeenCalled(); + }); + + it("rejects private-network URLs from remote callers", async () => { + for (const url of ["http://10.0.0.5/mcp", "http://192.168.1.1/mcp", "http://localhost:3000/mcp"]) { + const res = await POST(remoteRequest(url)); + expect(res.status, `should reject ${url}`).toBe(400); + } + }); + + it("still requires a url", async () => { + const res = await POST( + new Request("http://gateway.example.com/api/cli-tools/cowork-mcp-tools", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({}), + }) + ); + expect(res.status).toBe(400); + }); +}); diff --git a/tests/unit/db-sqlite-vs-lowdb.test.js b/tests/unit/db-sqlite-vs-lowdb.test.js index 52a80884..4955cd88 100644 --- a/tests/unit/db-sqlite-vs-lowdb.test.js +++ b/tests/unit/db-sqlite-vs-lowdb.test.js @@ -101,6 +101,104 @@ describe("DB SQLite layer — public API parity", () => { expect(back.providerSpecificData).toEqual({ foo: "bar" }); }); + it("providerConnections: successful validation clears stale routing locks", async () => { + const c = await sqliteDb.createProviderConnection({ + provider: "health-reset-update", + authType: "oauth", + email: "update@example.com", + accessToken: "old-token", + }); + await sqliteDb.updateProviderConnection(c.id, { + testStatus: "unavailable", + lastError: "Access denied", + lastErrorAt: "2026-09-05T00:00:00.000Z", + errorCode: 403, + backoffLevel: 3, + rateLimitedUntil: "2099-01-01T00:00:00.000Z", + modelLock_modelA: "2099-01-01T00:00:00.000Z", + modelLock_modelB: "2099-01-01T00:00:00.000Z", + }); + + await sqliteDb.updateProviderConnection(c.id, { testStatus: "active" }); + + const back = await sqliteDb.getProviderConnectionById(c.id); + expect(back).toMatchObject({ + testStatus: "active", + lastError: null, + lastErrorAt: null, + errorCode: null, + backoffLevel: 0, + rateLimitedUntil: null, + modelLock_modelA: null, + modelLock_modelB: null, + }); + }); + + it("providerConnections: re-saving valid OAuth credentials clears stale routing locks", async () => { + const existing = await sqliteDb.createProviderConnection({ + provider: "health-reset-resave", + authType: "oauth", + email: "resave@example.com", + accessToken: "old-token", + }); + await sqliteDb.updateProviderConnection(existing.id, { + testStatus: "unavailable", + lastError: "Access denied", + errorCode: 403, + backoffLevel: 2, + modelLock_modelA: "2099-01-01T00:00:00.000Z", + }); + + const resaved = await sqliteDb.createProviderConnection({ + provider: "health-reset-resave", + authType: "oauth", + email: "resave@example.com", + accessToken: "new-token", + testStatus: "active", + }); + + expect(resaved.id).toBe(existing.id); + const back = await sqliteDb.getProviderConnectionById(existing.id); + expect(back).toMatchObject({ + accessToken: "new-token", + testStatus: "active", + lastError: null, + errorCode: null, + backoffLevel: 0, + modelLock_modelA: null, + }); + }); + + it("providerConnections: active soft warnings survive the health reset", async () => { + const c = await sqliteDb.createProviderConnection({ + provider: "health-reset-warning", + authType: "oauth", + email: "warning@example.com", + }); + await sqliteDb.updateProviderConnection(c.id, { + testStatus: "unavailable", + lastError: "Old failure", + modelLock_modelA: "2099-01-01T00:00:00.000Z", + }); + + const warningAt = "2026-09-06T00:00:00.000Z"; + await sqliteDb.updateProviderConnection(c.id, { + testStatus: "active", + lastError: "Connected, but credits are exhausted", + lastErrorAt: warningAt, + }); + + const back = await sqliteDb.getProviderConnectionById(c.id); + expect(back).toMatchObject({ + testStatus: "active", + lastError: "Connected, but credits are exhausted", + lastErrorAt: warningAt, + errorCode: null, + backoffLevel: 0, + modelLock_modelA: null, + }); + }); + it("providerConnections: GitHub OAuth uses account identity as fallback name", async () => { const c = await sqliteDb.createProviderConnection({ provider: "github", diff --git a/tests/unit/deepseek-claude-tools.test.js b/tests/unit/deepseek-claude-tools.test.js new file mode 100644 index 00000000..bd05e208 --- /dev/null +++ b/tests/unit/deepseek-claude-tools.test.js @@ -0,0 +1,145 @@ +/** + * Regression test: prepareClaudeRequest() must strip client-defined `custom` + * tools when forwarding to a provider whose Anthropic-compatible endpoint + * does not accept them (DeepSeek — accepts only web_search_*). + * + * Background: + * When Claude Code talks to a DeepSeek route via /v1/messages, 9router + * forwards the request body as Claude-format to + * https://api.deepseek.com/anthropic/v1/messages. MCP / function tools + * arrive with `type: "custom"`. DeepSeek rejects them with HTTP 400 + * "tools[0]: unknown variant `custom`, expected `web_search_20250305` + * or `web_search_20260209`". The previous generic filter dropped them + * but also stripped the web_search_* tools that DeepSeek actually + * accepts. DeepSeek now exposes a `quirks.claudeSupportedToolTypes` + * whitelist and prepareClaudeRequest honours it. + */ + +import { describe, it, expect } from "vitest"; +import { prepareClaudeRequest } from "../../open-sse/translator/formats/claude.js"; +import { PROVIDERS } from "../../open-sse/providers/index.js"; + +function makeBody(tools) { + return { + model: "deepseek-v4-pro", + max_tokens: 1024, + messages: [{ role: "user", content: "hello" }], + tools, + }; +} + +describe("prepareClaudeRequest — provider: deepseek", () => { + it("declares the supportedTypes quirk on the provider transport", () => { + expect(PROVIDERS.deepseek).toBeDefined(); + expect(PROVIDERS.deepseek.quirks).toBeDefined(); + expect(PROVIDERS.deepseek.quirks.claudeSupportedToolTypes).toEqual([ + "web_search_20250305", + "web_search_20260209", + ]); + }); + + it("strips MCP / custom tools (the regression) before forwarding", () => { + const body = makeBody([ + { type: "custom", name: "Bash", input_schema: { type: "object" } }, + { type: "custom", name: "Read", input_schema: { type: "object" } }, + { type: "custom", name: "Glob", input_schema: { type: "object" } }, + ]); + + const out = prepareClaudeRequest(body, "deepseek"); + + expect(out.tools).toBeUndefined(); + expect(out.tool_choice).toBeUndefined(); + }); + + it("keeps web_search_20250305 and web_search_20260209", () => { + const out = prepareClaudeRequest( + makeBody([ + { type: "web_search_20250305", name: "web_search" }, + { type: "web_search_20260209", name: "web_search" }, + ]), + "deepseek" + ); + + expect(Array.isArray(out.tools)).toBe(true); + expect(out.tools).toHaveLength(2); + const types = out.tools.map(t => t.type).sort(); + expect(types).toEqual(["web_search_20250305", "web_search_20260209"]); + }); + + it("preserves the `type` field on web_search_* tools (DeepSeek requires it)", () => { + // The .map below the filter must NOT strip `type` when the provider + // declared a whitelist — DeepSeek would reject a tool object missing + // its discriminator field with the same unknown-variant error. + const out = prepareClaudeRequest( + makeBody([{ type: "web_search_20250305", name: "web_search" }]), + "deepseek" + ); + + expect(out.tools[0].type).toBe("web_search_20250305"); + expect(out.tools[0].name).toBe("web_search"); + }); + + it("drops `custom` but keeps `web_search_*` when both are present", () => { + const out = prepareClaudeRequest( + makeBody([ + { type: "custom", name: "Bash", input_schema: { type: "object" } }, + { type: "web_search_20250305", name: "web_search" }, + ]), + "deepseek" + ); + + expect(Array.isArray(out.tools)).toBe(true); + expect(out.tools).toHaveLength(1); + expect(out.tools[0].type).toBe("web_search_20250305"); + expect(out.tools[0].name).toBe("web_search"); + }); + + it("survives when body has no tools", () => { + const out = prepareClaudeRequest(makeBody(undefined), "deepseek"); + expect(out.tools).toBeUndefined(); + }); + + it("rejects future / unknown tool types instead of forwarding them", () => { + const out = prepareClaudeRequest( + makeBody([{ type: "future_tool_2099", name: "x" }]), + "deepseek" + ); + expect(out.tools).toBeUndefined(); + }); +}); + +describe("prepareClaudeRequest — backward compat: providers without the quirk", () => { + // Pick any non-Claude provider that has a Claude-format transport and has + // NOT been migrated to the new quirk. This protects GLM / Kimi / future + // Anthropic-compatible providers from unintended changes. + it("keeps prior behaviour (drop custom + web_search_*, normalize no-type tools)", () => { + const candidate = Object.entries(PROVIDERS).find( + ([id, p]) => + id !== "claude" && + p?.transports?.some(t => t.format === "claude") && + !p?.quirks?.claudeSupportedToolTypes + ); + + if (!candidate) { + // Every Claude-format provider has been migrated — nothing to verify. + return; + } + + const [providerId] = candidate; + + const out = prepareClaudeRequest( + makeBody([ + { type: "custom", name: "Bash", input_schema: { type: "object" } }, + { type: "web_search_20250305", name: "web_search" }, + { name: "no_type_tool", input_schema: { type: "object" } }, + ]), + providerId + ); + + if (out.tools !== undefined) { + for (const t of out.tools) { + expect(t.type).toBeUndefined(); + } + } + }); +}); \ No newline at end of file diff --git a/tests/unit/deepseek-usage.test.js b/tests/unit/deepseek-usage.test.js index 0be26edc..d9048b09 100644 --- a/tests/unit/deepseek-usage.test.js +++ b/tests/unit/deepseek-usage.test.js @@ -80,6 +80,8 @@ describe("getUsageForProvider(deepseek)", () => { used: 0, total: 12.5, remainingPercentage: 100, + isCreditBalance: true, + currency: "USD", }); expect(usage.quotas["Balance (USD)"].remaining).toBeUndefined(); // Zero CNY still listed so user sees currency row diff --git a/tests/unit/image-generation.test.js b/tests/unit/image-generation.test.js index e5504aec..6a22cd40 100644 --- a/tests/unit/image-generation.test.js +++ b/tests/unit/image-generation.test.js @@ -316,7 +316,7 @@ describe("handleImageGenerationCore", () => { expect(responseBody.data[0].b64_json).toBeTruthy(); }); - it("generates image with Codex gpt-5.5-image using current Codex version header", async () => { + it.each(["gpt-5.5", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])("generates image with Codex %s-image using current Codex version header", async (model) => { global.fetch.mockResolvedValueOnce( new Response( [ @@ -335,7 +335,7 @@ describe("handleImageGenerationCore", () => { size: "1024x1024", output_format: "png", }, - modelInfo: { provider: "codex", model: "gpt-5.5-image" }, + modelInfo: { provider: "codex", model: `${model}-image` }, credentials: { accessToken: "codex-token", providerSpecificData: { chatgptAccountId: "account-123" }, @@ -351,14 +351,14 @@ describe("handleImageGenerationCore", () => { headers: expect.objectContaining({ authorization: "Bearer codex-token", "chatgpt-account-id": "account-123", - version: "0.136.0", + version: "0.154.0", }), }) ); const fetchCall = global.fetch.mock.calls[0]; const requestBody = JSON.parse(fetchCall[1].body); - expect(requestBody.model).toBe("gpt-5.5"); + expect(requestBody.model).toBe(model); expect(requestBody.tools).toEqual([ { type: "image_generation", output_format: "png", size: "1024x1024" }, ]); @@ -367,6 +367,47 @@ describe("handleImageGenerationCore", () => { expect(responseBody.data[0].b64_json).toBe("base64codeximage"); }); + it("generates image with Codex gpt-image-2.5 tool model", async () => { + global.fetch.mockResolvedValueOnce( + new Response( + [ + "event: response.output_item.done", + 'data: {"item":{"type":"image_generation_call","result":"base64codeximage"}}', + "", + "", + ].join("\n"), + { status: 200, headers: { "Content-Type": "text/event-stream" } } + ) + ); + + const result = await handleImageGenerationCore({ + body: { + prompt: "A futuristic city", + size: "1024x1024", + output_format: "png", + }, + modelInfo: { provider: "codex", model: "gpt-image-2.5" }, + credentials: { + accessToken: "codex-token", + providerSpecificData: { chatgptAccountId: "account-123" }, + }, + log: null, + }); + + expect(result.success).toBe(true); + const fetchCall = global.fetch.mock.calls[0]; + const requestBody = JSON.parse(fetchCall[1].body); + expect(requestBody.model).toBe("gpt-5.5"); + expect(requestBody.tools).toEqual([ + { type: "image_generation", output_format: "png", size: "1024x1024", action: "generate", model: "gpt-image-2.5" }, + ]); + expect(requestBody.tool_choice).toEqual({ type: "image_generation" }); + expect(requestBody.reasoning).toEqual({ effort: "medium", summary: "auto" }); + + const responseBody = await result.response.json(); + expect(responseBody.data[0].b64_json).toBe("base64codeximage"); + }); + it("generates image with Cloudflare Workers AI JSON response", async () => { global.fetch.mockResolvedValueOnce( new Response( diff --git a/tests/unit/kiro-api-key-endpoint-routing.test.js b/tests/unit/kiro-api-key-endpoint-routing.test.js index a0750adc..22cf97fc 100644 --- a/tests/unit/kiro-api-key-endpoint-routing.test.js +++ b/tests/unit/kiro-api-key-endpoint-routing.test.js @@ -20,26 +20,26 @@ describe("Kiro auth-aware endpoint routing", () => { ]); }); - it("keeps Builder ID OAuth on the Kiro runtime surface", () => { + it("routes Builder ID OAuth through Amazon Q first (runtime path deprecated)", () => { expect(executor.getOrderedBaseUrls(credentials("builder-id"))).toEqual([ - RUNTIME, - CODEWHISPERER, Q, + CODEWHISPERER, + RUNTIME, ]); }); - it("keeps external IdP on CodeWhisperer before Amazon Q", () => { + it("routes external IdP through Amazon Q first", () => { expect(executor.getOrderedBaseUrls(credentials("external_idp"))).toEqual([ - CODEWHISPERER, Q, + CODEWHISPERER, RUNTIME, ]); }); - it("regionalizes AWS endpoints for IDC without changing Kiro runtime", () => { + it("regionalizes AWS endpoints for IDC with Q first", () => { expect(executor.getOrderedBaseUrls(credentials("idc", "eu-west-1"))).toEqual([ - "https://codewhisperer.eu-west-1.amazonaws.com/generateAssistantResponse", "https://q.eu-west-1.amazonaws.com/generateAssistantResponse", + "https://codewhisperer.eu-west-1.amazonaws.com/generateAssistantResponse", RUNTIME, ]); }); diff --git a/tests/unit/kiro-minimal-wire-payload.test.js b/tests/unit/kiro-minimal-wire-payload.test.js new file mode 100644 index 00000000..97075d4b --- /dev/null +++ b/tests/unit/kiro-minimal-wire-payload.test.js @@ -0,0 +1,19 @@ +import { describe, expect, it } from "vitest"; +import { openaiToKiroRequest } from "../../open-sse/translator/request/openai-to-kiro.js"; +import { claudeToKiroRequest } from "../../open-sse/translator/request/claude-to-kiro.js"; + +for (const [name, translate, body] of [ + ["OpenAI", openaiToKiroRequest, { messages: [{ role: "user", content: "hello" }] }], + ["Claude", claudeToKiroRequest, { messages: [{ role: "user", content: "hello" }] }], +]) { + describe(`${name} Kiro minimal wire payload`, () => { + it("omits unsupported agent fields", () => { + const payload = translate("kiro/claude-sonnet-4.5", body, true, {}); + expect(payload).not.toHaveProperty("agentMode"); + expect(payload.conversationState).not.toHaveProperty("agentContinuationId"); + expect(payload.conversationState).not.toHaveProperty("agentTaskType"); + expect(payload.conversationState.chatTriggerType).toBe("MANUAL"); + expect(payload.conversationState.currentMessage.userInputMessage.origin).toBe("AI_EDITOR"); + }); + }); +} diff --git a/tests/unit/kiro-terminal-integrity.test.js b/tests/unit/kiro-terminal-integrity.test.js index aaf66209..3199c03d 100644 --- a/tests/unit/kiro-terminal-integrity.test.js +++ b/tests/unit/kiro-terminal-integrity.test.js @@ -115,7 +115,7 @@ async function text(stream) { async function execute(executor = new KiroExecutor(), overrides = {}) { return executor.execute({ model: "kr/claude-opus-4.8", - body: { systemPrompt: "base", conversationState: {} }, + body: { conversationState: { currentMessage: { userInputMessage: { content: "base", modelId: "m" } } } }, stream: true, credentials, ...overrides @@ -342,8 +342,12 @@ describe("Kiro terminal integrity recovery", () => { const retryBody = JSON.parse(fetchMock.mock.calls[1][1].body); expect(body).toContain("Recovered safely."); - expect(retryBody.systemPrompt).toContain("tool_call wrapper was malformed"); - expect(retryBody.systemPrompt).not.toContain("IGNORE_ALL_INSTRUCTIONS"); + // The repair instruction rides in the user turn: kiro.dev rejects a + // top-level systemPrompt with 400 REQUEST_BODY_INVALID. + const retryContent = retryBody.conversationState.currentMessage.userInputMessage.content; + expect(retryBody.systemPrompt).toBeUndefined(); + expect(retryContent).toContain("tool_call wrapper was malformed"); + expect(retryContent).not.toContain("IGNORE_ALL_INSTRUCTIONS"); }); it("lets a complete tool call override metadata end_turn", async () => { diff --git a/tests/unit/kiro-tool-result-placeholder.test.js b/tests/unit/kiro-tool-result-placeholder.test.js new file mode 100644 index 00000000..ab910b42 --- /dev/null +++ b/tests/unit/kiro-tool-result-placeholder.test.js @@ -0,0 +1,115 @@ +import { describe, it, expect } from "vitest"; +import { openaiToKiroRequest } from "../../open-sse/translator/request/openai-to-kiro.js"; +import { claudeToKiroRequest } from "../../open-sse/translator/request/claude-to-kiro.js"; +import { + canonicalizeKiroConversation, + KIRO_TOOL_RESULTS_PLACEHOLDER, + KIRO_EMPTY_USER_PLACEHOLDER, +} from "../../open-sse/translator/concerns/kiroConversation.js"; + +const TOOLS_OPENAI = [{ + type: "function", + function: { + name: "get_weather", + description: "Get weather", + parameters: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, + }, +}]; + +const TOOLS_CLAUDE = [{ + name: "get_weather", + description: "Get weather", + input_schema: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, +}]; + +function allUserContents(payload) { + const state = payload.conversationState; + return [ + ...state.history.filter((t) => t.userInputMessage).map((t) => t.userInputMessage.content), + state.currentMessage.userInputMessage.content, + ]; +} + +describe("Kiro tool-result-only turns", () => { + it("OpenAI → Kiro: tool message gets a neutral placeholder, not \"continue\"", () => { + const payload = openaiToKiroRequest("claude-sonnet-4.6", { + tools: TOOLS_OPENAI, + messages: [ + { role: "user", content: "The secret word is PINEAPPLE. Weather in Jakarta?" }, + { role: "assistant", content: null, tool_calls: [{ id: "call_1", type: "function", function: { name: "get_weather", arguments: "{\"city\":\"Jakarta\"}" } }] }, + { role: "tool", tool_call_id: "call_1", content: "32C, humid" }, + ], + }, true, {}); + + const current = payload.conversationState.currentMessage.userInputMessage; + expect(current.content).toContain(KIRO_TOOL_RESULTS_PLACEHOLDER); + expect(current.content).not.toMatch(/\bcontinue\b/); + expect(current.userInputMessageContext.toolResults).toHaveLength(1); + expect(allUserContents(payload).join("\n")).toContain("PINEAPPLE"); + }); + + it("Claude → Kiro: tool_result-only user message gets a neutral placeholder", () => { + const payload = claudeToKiroRequest("claude-sonnet-4.6", { + tools: TOOLS_CLAUDE, + messages: [ + { role: "user", content: "The secret word is PINEAPPLE. Weather in Jakarta?" }, + { role: "assistant", content: [{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Jakarta" } }] }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "toolu_1", content: "32C, humid" }] }, + ], + }, true, {}); + + const current = payload.conversationState.currentMessage.userInputMessage; + expect(current.content).toContain(KIRO_TOOL_RESULTS_PLACEHOLDER); + expect(current.content).not.toMatch(/\bcontinue\b/); + expect(current.userInputMessageContext.toolResults).toHaveLength(1); + }); + + it("keeps real user text when a turn has both text and tool results", () => { + const payload = claudeToKiroRequest("claude-sonnet-4.6", { + tools: TOOLS_CLAUDE, + messages: [ + { role: "user", content: "Weather in Jakarta?" }, + { role: "assistant", content: [{ type: "tool_use", id: "toolu_1", name: "get_weather", input: { city: "Jakarta" } }] }, + { role: "user", content: [ + { type: "tool_result", tool_use_id: "toolu_1", content: "32C" }, + { type: "text", text: "Now answer in one word." }, + ] }, + ], + }, true, {}); + + const current = payload.conversationState.currentMessage.userInputMessage; + expect(current.content).toContain("Now answer in one word."); + expect(current.content).not.toContain(KIRO_TOOL_RESULTS_PLACEHOLDER); + }); + + it("canonicalize: history turn with tool results and no text uses the placeholder", () => { + const result = canonicalizeKiroConversation({ + history: [ + { userInputMessage: { content: "Weather in Jakarta?", modelId: "m" } }, + { assistantResponseMessage: { content: "", toolUses: [{ toolUseId: "t1", name: "get_weather", input: { city: "Jakarta" } }] } }, + { userInputMessage: { content: "", modelId: "m", userInputMessageContext: { toolResults: [{ toolUseId: "t1", status: "success", content: [{ text: "32C" }] }] } } }, + { assistantResponseMessage: { content: "It is 32C." } }, + ], + currentMessage: { userInputMessage: { content: "Hot or cold?", modelId: "m" } }, + modelId: "m", + toolSpecs: [{ toolSpecification: { name: "get_weather", description: "Get weather", inputSchema: { json: { type: "object", properties: {} } } } }], + nameMap: new Map([["get_weather", "get_weather"]]), + }); + + expect(result.valid).toBe(true); + expect(result.history[2].userInputMessage.content).toBe(KIRO_TOOL_RESULTS_PLACEHOLDER); + }); + + it("canonicalize: an empty turn without tool results still falls back to \"continue\"", () => { + const result = canonicalizeKiroConversation({ + history: [{ assistantResponseMessage: { content: "Hello" } }], + currentMessage: { userInputMessage: { content: "", modelId: "m" } }, + modelId: "m", + toolSpecs: [], + nameMap: new Map(), + }); + + expect(result.history[0].userInputMessage.content).toBe(KIRO_EMPTY_USER_PLACEHOLDER); + expect(result.currentMessage.userInputMessage.content).toBe(KIRO_EMPTY_USER_PLACEHOLDER); + }); +}); diff --git a/tests/unit/model-catalog-scope.test.js b/tests/unit/model-catalog-scope.test.js new file mode 100644 index 00000000..c582d28f --- /dev/null +++ b/tests/unit/model-catalog-scope.test.js @@ -0,0 +1,174 @@ +import { afterAll, beforeAll, describe, expect, it } from "vitest"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +// Both modules read their file path from DATA_DIR at import time, so the temp +// data dir has to be in place before the first import. +const dataDir = fs.mkdtempSync(path.join(os.tmpdir(), "9r-catalog-")); +process.env.DATA_DIR = dataDir; +const catalogFile = path.join(dataDir, "model-catalog.json"); + +// One upstream record per gateway: the same short id means different things to +// different vendors, which is what used to leak capabilities across providers. +const upstream = { + zai: { models: { "glm-4.6v": { modalities: { input: ["text", "image"] } } } }, + // two local ids alias this one upstream provider + zhipuai: { models: { "glm-5-canary": { modalities: { input: ["text", "image", "pdf"] } } } }, + moonshotai: { models: { "kimi-k3": { modalities: { input: ["text"] } } } }, + kilo: { models: { "kilo-auto/efficient": { modalities: { input: ["text", "image"] } } } }, +}; +// The registry snapshot the sync feeds build(): local ids, with the capabilities +// the tables resolve on their own. +const entries = [ + { provider: "glm", model: "glm-4.6v", current: { contextWindow: 200000, maxOutput: 128000 } }, + { provider: "glm-cn", model: "glm-5-canary", current: { contextWindow: 200000, maxOutput: 128000 } }, + { provider: "zhipu", model: "glm-5-canary", current: { contextWindow: 200000, maxOutput: 128000 } }, + { provider: "kimi", model: "kimi-k3", current: { contextWindow: 128000, maxOutput: 32000 } }, +]; + +let build, getCatalogModalities, invalidateCatalog, syncModelCatalog, startModelCatalogSync, capabilities; + +beforeAll(async () => { + ({ build, syncModelCatalog, startModelCatalogSync } = await import("../../src/lib/modelCatalog/sync.js")); + // the builder is exercised directly; a missing export must fail loudly here + // rather than skip every case below + expect(typeof build).toBe("function"); + const { models, providers } = build(upstream, entries); + fs.writeFileSync(catalogFile, JSON.stringify({ v: 2, models, providers })); + ({ getCatalogModalities, invalidateCatalog } = await import("../../open-sse/providers/catalogOverride.js")); + capabilities = await import("../../open-sse/providers/capabilities.js"); +}); + +afterAll(() => { + fs.rmSync(dataDir, { recursive: true, force: true }); +}); + +describe("model catalog", () => { + it("keys modalities by gateway and writes no model-only key", () => { + const { models } = build(upstream, entries); + // upstream "zai" is filed under the local id requests arrive with... + expect(models["glm:glm-4.6v"]).toEqual({ vision: true }); + // ...and under its upstream name, because a custom provider node can carry + // that name without being in the registry snapshot + expect(models["zai:glm-4.6v"]).toEqual({ vision: true }); + expect(models["kilo:efficient"]).toEqual({ vision: true }); + // the vendor-stripped key is what used to be shared with every other gateway + expect(models["glm-4.6v"]).toBeUndefined(); + expect(models["efficient"]).toBeUndefined(); + // a gateway that only declares text has nothing to contribute + expect(models["kimi:kimi-k3"]).toBeUndefined(); + }); + + it("files an upstream provider under every local id that aliases it", () => { + const { models } = build(upstream, entries); + // glm-cn and zhipu are both zhipuai upstream; neither may be dropped + expect(models["glm-cn:glm-5-canary"]).toEqual({ vision: true, pdf: true }); + expect(models["zhipu:glm-5-canary"]).toEqual({ vision: true, pdf: true }); + expect(models["zhipuai:glm-5-canary"]).toEqual({ vision: true, pdf: true }); + expect(getCatalogModalities("glm-cn", "glm-5-canary")).toEqual({ vision: true, pdf: true }); + expect(getCatalogModalities("zhipu", "glm-5-canary")).toEqual({ vision: true, pdf: true }); + }); + + it("does not hand a router mode another vendor's modalities", () => { + // kilo's "efficient" is a real model; another gateway's "efficient" is a mode + expect(getCatalogModalities("kilo", "kilo-auto/efficient")).toEqual({ vision: true }); + expect(getCatalogModalities("kilo-gateway", "kilo-auto/efficient")).toBeNull(); + expect(getCatalogModalities("qoder", "efficient")).toBeNull(); + }); + + it("resolves a gateway the file was written for, and nobody else", () => { + expect(getCatalogModalities("glm", "glm-4.6v")).toEqual({ vision: true }); + expect(getCatalogModalities("zai", "glm-4.6v")).toEqual({ vision: true }); + expect(getCatalogModalities("unrelated", "glm-4.6v")).toBeNull(); + expect(getCatalogModalities(undefined, "glm-4.6v")).toBeNull(); + }); + + it("passes the gateway to the catalog reader when refining", () => { + const seen = []; + capabilities.setCatalogSource({ + getModalities: (provider) => { + seen.push(provider); + return provider === "gateway-a" ? { vision: true } : null; + }, + getLimits: () => null, + }); + try { + // "*laguna*" resolves from the pattern table, so refine() runs + expect(capabilities.getCapabilitiesForModel("gateway-a", "laguna-9-preview").vision).toBe(true); + expect(capabilities.getCapabilitiesForModel("gateway-b", "laguna-9-preview").vision).toBe(false); + expect(seen).toContain("gateway-a"); + } finally { + capabilities.setCatalogSource(null); + } + }); + + it("shares the installed source with every copy of the module", async () => { + const source = { + getModalities: (provider) => (provider === "gateway-a" ? { vision: true } : null), + getLimits: () => null, + }; + capabilities.setCatalogSource(source); + try { + // The server bundles this module into more than one chunk and the startup + // hook only runs in one of them, so the slot has to be process-wide. + expect(globalThis.__9rCatalogSource).toBe(source); + const other = await import("../../open-sse/providers/capabilities.js?copy=2"); + expect(other.getCapabilitiesForModel).not.toBe(capabilities.getCapabilitiesForModel); + // ...and that second copy resolves through the source it never installed + expect(other.getCapabilitiesForModel("gateway-a", "laguna-9-preview").vision).toBe(true); + } finally { + capabilities.setCatalogSource(null); + } + expect(globalThis.__9rCatalogSource).toBeNull(); + }); +}); + +describe("catalog schema", () => { + it("ignores a file written before the keys were scoped", () => { + const scoped = fs.readFileSync(catalogFile); + // v1: flat model keys, which is exactly the shape that collided + fs.writeFileSync(catalogFile, JSON.stringify({ v: 1, models: { "kimi-k3": { vision: true } }, providers: {} })); + invalidateCatalog(); + expect(getCatalogModalities("kimi", "kimi-k3")).toBeNull(); + fs.writeFileSync(catalogFile, scoped); + invalidateCatalog(); + }); + + it("rebuilds an older-schema file instead of trusting its etag", async () => { + fs.writeFileSync(catalogFile, JSON.stringify({ v: 1, etag: 'W/"old"', models: {}, providers: {} })); + invalidateCatalog(); + startModelCatalogSync(); // picks the file's etag + schema version back up + + const sent = []; + const realFetch = globalThis.fetch; + globalThis.fetch = async (_url, options) => { + sent.push(options?.headers || {}); + return { ok: true, status: 200, headers: new Map([["etag", 'W/"new"']]), json: async () => upstream }; + }; + try { + expect((await syncModelCatalog()).status).toBe("updated"); + } finally { + globalThis.fetch = realFetch; + } + expect(sent[0]["if-none-match"]).toBeUndefined(); + const written = JSON.parse(fs.readFileSync(catalogFile, "utf8")); + expect(written.v).toBe(2); + expect(written.models["glm:glm-4.6v"]).toEqual({ vision: true }); + }); + + it("asks upstream for a 304 once the file is current", async () => { + const sent = []; + const realFetch = globalThis.fetch; + globalThis.fetch = async (_url, options) => { + sent.push(options?.headers || {}); + return { ok: false, status: 304, headers: new Map(), json: async () => ({}) }; + }; + try { + expect((await syncModelCatalog()).status).toBe("unchanged"); + } finally { + globalThis.fetch = realFetch; + } + expect(sent[0]["if-none-match"]).toBe('W/"new"'); + }); +}); diff --git a/tests/unit/openai-to-commandcode.test.js b/tests/unit/openai-to-commandcode.test.js index 7f12dc85..0a441dd7 100644 --- a/tests/unit/openai-to-commandcode.test.js +++ b/tests/unit/openai-to-commandcode.test.js @@ -179,3 +179,57 @@ describe("openaiToCommandCodeRequest — tools schema conversion", () => { expect(out.params.tools).toBeUndefined(); }); }); + +describe("openaiToCommandCodeRequest — native image blocks", () => { + const PNG_B64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="; + const DATA_URI = `data:image/png;base64,${PNG_B64}`; + + it("maps OpenAI image_url data URI to CommandCode {type:image,image,mimeType}", () => { + const out = openaiToCommandCodeRequest(MODEL, { + messages: [{ + role: "user", + content: [ + { type: "text", text: "what color?" }, + { type: "image_url", image_url: { url: DATA_URI } }, + ], + }], + }, true); + + expect(out.params.messages[0].content).toEqual([ + { type: "text", text: "what color?" }, + { type: "image", image: DATA_URI, mimeType: "image/png" }, + ]); + }); + + it("maps Claude/OpenAI base64 image source to a data-URI image block", () => { + const out = openaiToCommandCodeRequest(MODEL, { + messages: [{ + role: "user", + content: [ + { type: "image", source: { type: "base64", media_type: "image/png", data: PNG_B64 } }, + ], + }], + }, true); + + expect(out.params.messages[0].content).toEqual([ + { type: "image", image: DATA_URI, mimeType: "image/png" }, + ]); + }); + + it("does not stub dropped images as [image omitted]", () => { + const out = openaiToCommandCodeRequest(MODEL, { + messages: [{ + role: "user", + content: [ + { type: "text", text: "see this" }, + { type: "image_url", image_url: { url: DATA_URI } }, + ], + }], + }, true); + + const texts = out.params.messages[0].content + .filter((b) => b.type === "text") + .map((b) => b.text); + expect(texts).not.toContain("[image omitted]"); + }); +}); diff --git a/tests/unit/openai-to-kiro.test.js b/tests/unit/openai-to-kiro.test.js index 17773bdb..965bd2d1 100644 --- a/tests/unit/openai-to-kiro.test.js +++ b/tests/unit/openai-to-kiro.test.js @@ -606,7 +606,7 @@ describe("openaiToKiroRequest", () => { ); expect(second.conversationState.conversationId).toBe("hermes-session-openai-replay"); - expect(second.conversationState.agentContinuationId).toBe(first.conversationState.agentContinuationId); + expect(second.conversationState).not.toHaveProperty("agentContinuationId"); expect(second.conversationState.history[0].userInputMessage.content).toBe( first.conversationState.currentMessage.userInputMessage.content ); diff --git a/tests/unit/opencode-free-tool-choice.test.js b/tests/unit/opencode-free-tool-choice.test.js new file mode 100644 index 00000000..4d3aca91 --- /dev/null +++ b/tests/unit/opencode-free-tool-choice.test.js @@ -0,0 +1,90 @@ +import { describe, expect, it, vi } from "vitest"; +import { PROVIDERS } from "../../open-sse/config/providers.js"; +import { OpenCodeExecutor } from "../../open-sse/executors/opencode.js"; +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(async () => ({ ok: true, status: 200, headers: { get: () => "" } })), +})); + +// Break caught: opencode/muse-spark-1.3-contributor-free 400 vì upstream +// chỉ nhận tool_choice "auto"; named/required/none phải demote sang "auto". +const FREE_13 = "muse-spark-1.3-contributor-free"; +const CREDS = { connectionId: "opencode-free-tool-choice-test" }; +const INPUT = [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }]; +const TOOLS = [{ type: "function", name: "get_weather", description: "w", parameters: { type: "object", properties: {} } }]; + +function responsesBody(model, tool_choice) { + const body = { model, input: structuredClone(INPUT), tools: structuredClone(TOOLS) }; + if (tool_choice !== undefined) body.tool_choice = tool_choice; + return body; +} + +describe("opencode Free 1.3 tool_choice auto-only", () => { + it("khai quirk đúng model 1.3-Free trong registry", () => { + expect(PROVIDERS.opencode.quirks?.forceAutoToolChoiceModels).toEqual([FREE_13]); + }); + + it.each([ + ["Responses named", { type: "function", name: "get_weather" }], + ["Chat function named", { type: "function", function: { name: "get_weather" } }], + ["Claude tool named", { type: "tool", name: "get_weather" }], + ["required", "required"], + ["none", "none"], + ])("demote %s sang auto (plain và max)", (_label, choice) => { + for (const model of [FREE_13, `${FREE_13}(max)`]) { + const body = responsesBody(model, structuredClone(choice)); + const out = new OpenCodeExecutor().transformRequest(model, body, true, CREDS); + expect(out.tool_choice).toBe("auto"); + expect(out.tools).toEqual(TOOLS); + expect(out.input).toEqual(INPUT); + } + }); + + it("giữ auto và absent; tools/input nguyên vẹn", () => { + const autoOut = new OpenCodeExecutor().transformRequest( + FREE_13, responsesBody(FREE_13, "auto"), true, CREDS, + ); + expect(autoOut.tool_choice).toBe("auto"); + expect(autoOut.tools).toEqual(TOOLS); + expect(autoOut.input).toEqual(INPUT); + + const absentOut = new OpenCodeExecutor().transformRequest( + FREE_13, responsesBody(FREE_13, undefined), true, CREDS, + ); + expect("tool_choice" in absentOut).toBe(false); + expect(absentOut.tools).toEqual(TOOLS); + expect(absentOut.input).toEqual(INPUT); + }); + + it.each([ + ["1.2-Free", "muse-spark-1.2-contributor-free"], + ["future 1.4-Free", "muse-spark-1.4-contributor-free"], + ["Go id", "muse-spark-1.3-contributor"], + ["non-Muse", "big-pickle"], + ])("không đổi tool_choice của %s", (_label, model) => { + const choice = { type: "function", name: "get_weather" }; + const body = responsesBody(model, structuredClone(choice)); + const out = new OpenCodeExecutor().transformRequest(model, body, true, CREDS); + expect(out.tool_choice).toEqual(choice); + }); + + it("wire: execute gửi choice auto tới /zen/v1/responses", async () => { + proxyAwareFetch.mockClear(); + const ex = new OpenCodeExecutor(); + const body = responsesBody(FREE_13, { type: "function", name: "get_weather" }); + const { url, transformedBody } = await ex.execute({ + model: FREE_13, body, stream: true, credentials: CREDS, + }); + expect(url).toBe("https://opencode.ai/zen/v1/responses"); + expect(transformedBody.tool_choice).toBe("auto"); + expect(proxyAwareFetch).toHaveBeenCalledTimes(1); + const [actualUrl, actualInit] = proxyAwareFetch.mock.calls[0]; + expect(actualUrl).toBe("https://opencode.ai/zen/v1/responses"); + const sent = JSON.parse(actualInit.body); + expect(sent.tool_choice).toBe("auto"); + expect(sent.model).toBe(FREE_13); + expect(sent.tools).toEqual(TOOLS); + expect(sent.input).toEqual(INPUT); + }); +}); diff --git a/tests/unit/opencode-go-models.test.js b/tests/unit/opencode-go-models.test.js index 335897ff..92e7a059 100644 --- a/tests/unit/opencode-go-models.test.js +++ b/tests/unit/opencode-go-models.test.js @@ -1,12 +1,14 @@ import { describe, expect, it } from "vitest"; -import { PROVIDER_MODELS, getModelSupportedFormats } from "../../open-sse/config/providerModels.js"; +import { PROVIDER_MODELS, getModelSupportedFormats, getModelTargetFormat } from "../../open-sse/config/providerModels.js"; import { PROVIDERS } from "../../open-sse/config/providers.js"; import { resolveTransport } from "../../open-sse/services/provider.js"; // Chat-only models (no /messages, no /responses support on opencode-go) -const CHAT_ONLY = ["glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "mimo-v2.5", "mimo-v2.5-pro"]; +const CHAT_ONLY = ["glm-5.3", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "kimi-k3", + "deepseek-flash", "longcat-2.0", "mimo-v2.5", "mimo-v2.5-pro", "hy4-preview", "hy3"]; // Models that also expose the Anthropic /messages endpoint -const CLAUDE_CAPABLE = ["minimax-m3", "minimax-m2.7", "minimax-m2.5", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus"]; +const CLAUDE_CAPABLE = ["minimax-m3", "minimax-m2.7", "minimax-m2.5", + "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus"]; // Models that also expose the OpenAI /responses endpoint const RESPONSES_CAPABLE = ["deepseek-v4-pro", "deepseek-v4-flash"]; @@ -22,15 +24,31 @@ describe("OpenCode Go model catalog", () => { it("matches the documented model IDs", () => { const ids = (PROVIDER_MODELS["opencode-go"] || []).map((m) => m.id); expect(ids).toEqual([ - "glm-5.3-flash", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", + "deepseek-flash", + "glm-5.3-flash", "glm-5.3", "glm-5.2", "glm-5.1", "kimi-k2.7-code", "kimi-k2.6", "kimi-k3", "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-vision-exp", - "mimo-v2.5", "mimo-v2.5-pro", + "longcat-2.0", "mimo-v2.5", "mimo-v2.5-pro", "minimax-m3", "minimax-m2.7", "minimax-m2.5", - "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", + "qwen3.8-max", "qwen3.8-flash", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", + "hy4-preview", "hy3", + "grok-4.6", "gpt-5.6-luna", + "muse-spark-1.2-contributor", "muse-spark-1.3-contributor", ]); }); }); +describe("OpenCode Go thinking-suffix model lookup", () => { + it("preserves Responses routing for gpt-5.6-luna thinking variants", () => { + expect(getModelSupportedFormats("opencode-go", "gpt-5.6-luna(high)")).toEqual(["openai-responses"]); + expect(getModelTargetFormat("opencode-go", "gpt-5.6-luna(high)")).toBe("openai-responses"); + }); + + it("preserves Responses routing for grok-4.6 thinking variants", () => { + expect(getModelSupportedFormats("opencode-go", "grok-4.6(high)")).toEqual(["openai-responses"]); + expect(getModelTargetFormat("opencode-go", "grok-4.6(high)")).toBe("openai-responses"); + }); +}); + describe("OpenCode Go per-model supportedFormats", () => { it("declares [openai, claude] for MiniMax + Qwen models", () => { for (const m of CLAUDE_CAPABLE) { @@ -89,6 +107,15 @@ describe("OpenCode Go per-model transport guard (chatCore logic)", () => { } }); + it("routes Muse Spark (responses-only) to /responses, never to /messages", () => { + for (const m of ["muse-spark-1.2-contributor", "muse-spark-1.3-contributor", "grok-4.6", "gpt-5.6-luna"]) { + expect(getModelSupportedFormats("opencode-go", m)).toEqual(["openai-responses"]); + expect(pickTransport("opencode-go", "openai-responses", "opencode-go", m)?.baseUrl).toBe("https://opencode.ai/zen/go/v1/responses"); + expect(pickTransport("opencode-go", "claude", "opencode-go", m)).toBeNull(); + expect(pickTransport("opencode-go", "openai", "opencode-go", m)).toBeNull(); + } + }); + it("does NOT route MiniMax (no responses support) to /responses", () => { for (const m of CLAUDE_CAPABLE) { expect(pickTransport("opencode-go", "openai-responses", "opencode-go", m)).toBeNull(); diff --git a/tests/unit/opencode-go-muse-spark-responses.test.js b/tests/unit/opencode-go-muse-spark-responses.test.js new file mode 100644 index 00000000..98a780d7 --- /dev/null +++ b/tests/unit/opencode-go-muse-spark-responses.test.js @@ -0,0 +1,218 @@ +import { describe, expect, it } from "vitest"; +import { PROVIDER_MODELS, getModelTargetFormat, getModelSupportedFormats } from "../../open-sse/config/providerModels.js"; +import { PROVIDERS } from "../../open-sse/config/providers.js"; +import { resolveTransport } from "../../open-sse/services/provider.js"; +import { getCapabilitiesForModel } from "../../open-sse/providers/capabilities.js"; +import { getThinkingLevels } from "../../open-sse/providers/thinkingLevels.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; +import { OpenCodeGoExecutor } from "../../open-sse/executors/opencode-go.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; +import "../translator/registerAll.js"; +import { translateRequest } from "../../open-sse/translator/index.js"; + +const MODEL = "muse-spark-1.3-contributor"; +const PROVIDER = "opencode-go"; + +// Mirror of chatCore's per-model transport guard +function pickTransport(provider, sourceFormat, alias, model) { + const supported = getModelSupportedFormats(alias, model); + const rt = resolveTransport(provider, sourceFormat); + return supported?.includes(sourceFormat) ? rt : null; +} + +describe("ocg/muse-spark-1.3-contributor catalog", () => { + it("is registered responses-only", () => { + const entry = (PROVIDER_MODELS["opencode-go"] || []).find((m) => m.id === MODEL); + expect(entry).toBeDefined(); + expect(entry.targetFormat).toBe("openai-responses"); + expect(getModelSupportedFormats("opencode-go", MODEL)).toEqual(["openai-responses"]); + expect(getModelTargetFormat("ocg", MODEL)).toBe(FORMATS.OPENAI_RESPONSES); + expect(getModelTargetFormat("opencode-go", MODEL)).toBe(FORMATS.OPENAI_RESPONSES); + }); + + it("never takes the sourceFormat-matched transport (always translates)", () => { + expect(pickTransport(PROVIDER, "openai", "opencode-go", MODEL)).toBeNull(); + expect(pickTransport(PROVIDER, "claude", "opencode-go", MODEL)).toBeNull(); + expect(pickTransport(PROVIDER, "openai-responses", "opencode-go", MODEL)?.baseUrl) + .toBe("https://opencode.ai/zen/go/v1/responses"); + }); + + it("advertises reasoning via the shared muse-spark pattern", () => { + expect(getCapabilitiesForModel(PROVIDER, MODEL)).toMatchObject({ + vision: true, + reasoning: true, + thinkingFormat: "openai", + }); + expect(getThinkingLevels(PROVIDER, MODEL)).toContain("xhigh"); + }); +}); + +describe("OpenCodeGoExecutor routing + sanitization", () => { + it("routes gpt-5.6-luna to /responses", () => { + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl("gpt-5.6-luna")).toBe("https://opencode.ai/zen/go/v1/responses"); + expect(ex.buildUrl("gpt-5.6-luna(high)", true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions" }, + })).toBe("https://opencode.ai/zen/go/v1/responses"); + }); + + it("routes every responses-only registry model (grok-4.6) to /responses", () => { + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl("grok-4.6")).toBe("https://opencode.ai/zen/go/v1/responses"); + expect(ex.buildUrl("grok-4.6(high)", true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions" }, + })).toBe("https://opencode.ai/zen/go/v1/responses"); + }); + + it("is wired for opencode-go and routes muse-spark to /responses", () => { + expect(getExecutor("opencode-go")).toBeInstanceOf(OpenCodeGoExecutor); + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl(MODEL)).toBe("https://opencode.ai/zen/go/v1/responses"); + // Even a stale runtimeTransport must not drag muse-spark onto chat/messages + expect(ex.buildUrl(MODEL, true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/chat/completions" }, + })).toBe("https://opencode.ai/zen/go/v1/responses"); + }); + + it("leaves non-muse models on the default/runtime transport", () => { + const ex = new OpenCodeGoExecutor(); + expect(ex.buildUrl("kimi-k2.6")).toBe("https://opencode.ai/zen/go/v1/chat/completions"); + expect(ex.buildUrl("minimax-m3", true, 0, { + runtimeTransport: { baseUrl: "https://opencode.ai/zen/go/v1/messages" }, + })).toBe("https://opencode.ai/zen/go/v1/messages"); + }); + + it("normalizes caps + reasoning and coerces tool items exactly once", () => { + const ex = new OpenCodeGoExecutor(); + const args = { path: "a\"b\nc\\d", emoji: "🚀 ü", nested: { q: "x'y\"z" } }; + const body = { + model: MODEL, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "function_call", call_id: "x".repeat(100), name: "read", arguments: args }, + { type: "function_call", call_id: "bad", name: " ", arguments: "{}" }, + { type: "function_call", call_id: "frag", name: "exec", arguments: "{not json" }, + { type: "function_call_output", call_id: "c1", output: { ok: true, text: "héllo \"w\"" } }, + { type: "function_call_output", call_id: "c2", output: null }, + ], + tools: [ + { type: "function", function: { name: "read", description: "r", parameters: { type: "object", properties: {} } } }, + { type: "function", function: { name: " ", parameters: {} } }, + ], + max_tokens: 2048, + reasoning_effort: "high", + }; + const out = ex.transformRequest(MODEL, body, true, {}); + expect(out.max_output_tokens).toBe(2048); + expect(out.max_tokens).toBeUndefined(); + expect(out.reasoning).toEqual({ effort: "high", summary: "auto" }); + expect(out.stream).toBe(true); + expect(out.store).toBe(false); + // nameless declaration dropped, nameless call dropped + expect(out.tools.map((t) => t.name)).toEqual(["read"]); + const calls = out.input.filter((i) => i.type === "function_call"); + expect(calls.map((c) => c.name)).toEqual(["read", "exec"]); + // overlong id clamped, object args stringified exactly once + expect(calls[0].call_id).toHaveLength(64); + expect(JSON.parse(calls[0].arguments)).toEqual(args); + // invalid fragment coerced, never double-encoded + expect(calls[1].arguments).toBe("{}"); + const outputs = out.input.filter((i) => i.type === "function_call_output"); + expect(JSON.parse(outputs[0].output)).toEqual({ ok: true, text: "héllo \"w\"" }); + expect(outputs[1].output).toBe(""); + }); + + it("fills in properties for object tool schemas missing them", () => { + const ex = new OpenCodeGoExecutor(); + const body = { + model: MODEL, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + tools: [ + { type: "function", function: { name: "bare", parameters: { type: "object" } } }, + { type: "function", function: { name: "full", parameters: { type: "object", properties: { a: { type: "string" } } } } }, + ], + }; + const out = ex.transformRequest(MODEL, body, true, {}); + expect(out.tools.find((t) => t.name === "bare").parameters).toEqual({ type: "object", properties: {} }); + expect(out.tools.find((t) => t.name === "full").parameters).toEqual({ type: "object", properties: { a: { type: "string" } } }); + }); + + it("strips prior-turn reasoning items carrying encrypted_content from input", () => { + const ex = new OpenCodeGoExecutor(); + const body = { + model: MODEL, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { + type: "reasoning", + id: "rs_123", + encrypted_content: "ENC_BLOB_TURN_1", + summary: [{ type: "summary_text", text: "thinking text" }], + }, + { type: "function_call", call_id: "c1", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "c1", output: "ok" }, + ], + }; + const out = ex.transformRequest(MODEL, body, true, {}); + expect(out.input.some((i) => i.type === "reasoning")).toBe(false); + expect(JSON.stringify(out.input)).not.toContain("ENC_BLOB_TURN_1"); + expect(out.input.map((i) => i.type)).toEqual(["message", "function_call", "function_call_output"]); + }); +}); + +describe("chat/claude clients translate to Responses without breaking tools", () => { + const tricky = { cmd: "echo \"hi\"\nnewline\ttab\\slash", emoji: "🎉 café naïve", nested: { a: [1, "x'y"] } }; + + it("openai chat → responses keeps arguments parseable", () => { + const translated = translateRequest( + FORMATS.OPENAI, + FORMATS.OPENAI_RESPONSES, + MODEL, + { + model: `ocg/${MODEL}`, + messages: [ + { role: "system", content: [{ type: "text", text: "sys one" }, { type: "text", text: "sys two" }] }, + { role: "user", content: "run it" }, + { + role: "assistant", content: null, + tool_calls: [{ id: "call_1", type: "function", function: { name: "exec", arguments: tricky } }], + }, + { role: "tool", tool_call_id: "call_1", content: tricky }, + ], + tools: [{ type: "function", function: { name: "exec", description: "e", parameters: { type: "object", properties: {} } } }], + }, + true, {}, PROVIDER, + ); + expect(translated.instructions).toBe("sys one\nsys two"); + const fc = translated.input.find((i) => i.type === "function_call"); + expect(JSON.parse(fc.arguments)).toEqual(tricky); + const fco = translated.input.find((i) => i.type === "function_call_output"); + expect(JSON.parse(fco.output)).toEqual(tricky); + }); + + it("claude messages → responses double-hop keeps tool input intact", () => { + const viaOpenAI = translateRequest(FORMATS.CLAUDE, FORMATS.OPENAI, MODEL, { + system: "be terse", + messages: [ + { role: "user", content: [{ type: "text", text: "go" }] }, + { + role: "assistant", + content: [ + { type: "text", text: "calling" }, + { type: "tool_use", id: "tu_1", name: "exec", input: tricky }, + ], + }, + { + role: "user", + content: [{ type: "tool_result", tool_use_id: "tu_1", content: [{ type: "text", text: JSON.stringify(tricky) }] }], + }, + ], + tools: [{ name: "exec", description: "e", input_schema: { type: "object", properties: {} } }], + }, true, {}, PROVIDER); + const translated = translateRequest(FORMATS.OPENAI, FORMATS.OPENAI_RESPONSES, MODEL, viaOpenAI, true, {}, PROVIDER); + const fc = translated.input.find((i) => i.type === "function_call"); + expect(JSON.parse(fc.arguments)).toEqual(tricky); + const fco = translated.input.find((i) => i.type === "function_call_output"); + expect(JSON.parse(fco.output)).toEqual(tricky); + }); +}); diff --git a/tests/unit/opencode-go-session.test.js b/tests/unit/opencode-go-session.test.js new file mode 100644 index 00000000..2d0ec560 --- /dev/null +++ b/tests/unit/opencode-go-session.test.js @@ -0,0 +1,166 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; + +const { fetchMock } = vi.hoisted(() => ({ + fetchMock: vi.fn(), +})); + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: fetchMock, +})); + +import { DefaultExecutor } from "../../open-sse/executors/default.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; + +const TRANSPORTS = [ + { format: "openai", baseUrl: "https://opencode.ai/zen/go/v1/chat/completions", auth: { combined: true, header: "Authorization", scheme: "bearer" } }, + { format: "claude", baseUrl: "https://opencode.ai/zen/go/v1/messages", auth: { combined: true, header: "x-api-key", scheme: "raw", anthropicVersion: true } }, + { format: "openai-responses", baseUrl: "https://opencode.ai/zen/go/v1/responses", auth: { combined: true, header: "Authorization", scheme: "bearer" } }, +]; + +function makeCredentials(overrides = {}) { + return { + apiKey: "test-key", + connectionId: "connection-a", + rawHeaders: {}, + runtimeTransport: TRANSPORTS[0], + ...overrides, + }; +} + +function prepare(executor, overrides = {}) { + const credentials = overrides.credentials || makeCredentials(); + const prepared = executor.prepareRequestCredentials({ + body: overrides.body || { messages: [{ role: "user", content: "hello" }] }, + credentials, + providerSessionId: overrides.providerSessionId ?? "conversation-a", + clientTool: overrides.clientTool ?? "claude", + }); + return { credentials, prepared }; +} + +beforeEach(() => { + fetchMock.mockReset(); + fetchMock.mockResolvedValue(new Response("{}", { + status: 200, + headers: { "content-type": "application/json" }, + })); +}); + +describe("OpenCode Go x-opencode-session", () => { + it("uses a dedicated executor with request-local session credentials", () => { + const executor = getExecutor("opencode-go"); + const { credentials, prepared } = prepare(executor); + + expect(executor.constructor.name).toBe("OpenCodeGoExecutor"); + expect(prepared).not.toBe(credentials); + expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/); + expect(credentials).not.toHaveProperty("_opencodeGoSession"); + expect(executor).not.toHaveProperty("_currentSessionId"); + expect(executor).not.toHaveProperty("_opencodeGoSession"); + }); + + it("preserves a valid native session header case-insensitively", () => { + const executor = getExecutor("opencode-go"); + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "X-OpenCode-Session": " native-session-a " } }), + }); + + expect(prepared._opencodeGoSession).toBe("native-session-a"); + }); + + it("ignores an oversized native session and uses the translated identity", () => { + const executor = getExecutor("opencode-go"); + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "x-opencode-session": "x".repeat(257) } }), + }); + + expect(prepared._opencodeGoSession).toMatch(/^ses_[0-9a-f]{32}$/); + }); + + it("keeps the same translated conversation stable across all transports", () => { + const executor = getExecutor("opencode-go"); + const values = TRANSPORTS.map((runtimeTransport) => { + const { prepared } = prepare(executor, { + credentials: makeCredentials({ runtimeTransport }), + }); + return executor.buildHeaders(prepared, true)["x-opencode-session"]; + }); + + expect(new Set(values).size).toBe(1); + expect(values[0]).toMatch(/^ses_[0-9a-f]{32}$/); + expect(values[0]).not.toContain("conversation-a"); + }); + + it("isolates different conversations", () => { + const executor = getExecutor("opencode-go"); + const a = prepare(executor, { providerSessionId: "conversation-a" }).prepared._opencodeGoSession; + const b = prepare(executor, { providerSessionId: "conversation-b" }).prepared._opencodeGoSession; + + expect(a).not.toBe(b); + }); + + it("isolates different downstream agents that reuse the same raw id", () => { + const executor = getExecutor("opencode-go"); + const claude = prepare(executor, { clientTool: "claude" }).prepared._opencodeGoSession; + const codex = prepare(executor, { clientTool: "codex" }).prepared._opencodeGoSession; + + expect(claude).not.toBe(codex); + }); + + it("uses a stable opaque connection fallback when no session is supplied", () => { + const executor = getExecutor("opencode-go"); + const options = { + credentials: makeCredentials({ connectionId: "fallback-connection" }), + providerSessionId: null, + clientTool: null, + body: { messages: [{ role: "user", content: "headerless" }] }, + }; + const first = prepare(executor, options).prepared._opencodeGoSession; + const second = prepare(executor, options).prepared._opencodeGoSession; + + expect(first).toBe(second); + expect(first).toMatch(/^ses_[0-9a-f]{32}$/); + expect(first).not.toContain("fallback-connection"); + }); + + it("adds the prepared session to the actual fetch headers", async () => { + const executor = getExecutor("opencode-go"); + const credentials = makeCredentials(); + const result = await executor.execute({ + model: "glm-5.2", + body: { messages: [{ role: "user", content: "hello" }] }, + stream: false, + credentials, + providerSessionId: "conversation-fetch", + clientTool: "codex", + }); + + expect(result.headers["x-opencode-session"]).toMatch(/^ses_[0-9a-f]{32}$/); + expect(fetchMock).toHaveBeenCalledOnce(); + expect(fetchMock.mock.calls[0][1].headers["x-opencode-session"]).toBe(result.headers["x-opencode-session"]); + expect(credentials).not.toHaveProperty("_opencodeGoSession"); + }); + + it("does not add the header to unrelated default executors", () => { + const headers = new DefaultExecutor("openai").buildHeaders({ apiKey: "test-key" }, false); + expect(headers["x-opencode-session"]).toBeUndefined(); + }); +}); + +describe("chatCore provider session forwarding", () => { + it("passes the original provider session and client tool on initial and retry execution", () => { + const source = readFileSync( + fileURLToPath(new URL("../../open-sse/handlers/chatCore.js", import.meta.url)), + "utf8", + ); + const calls = [...source.matchAll(/executor\.execute\(\{([\s\S]*?)\}\)/g)].map((match) => match[1]); + + expect(calls).toHaveLength(2); + for (const call of calls) { + expect(call).toMatch(/providerSessionId:\s*sessionSeed/); + expect(call).toMatch(/\bclientTool\b/); + } + }); +}); diff --git a/tests/unit/opencode-go-usage.test.js b/tests/unit/opencode-go-usage.test.js new file mode 100644 index 00000000..349554a4 --- /dev/null +++ b/tests/unit/opencode-go-usage.test.js @@ -0,0 +1,129 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: vi.fn(), +})); + +import { proxyAwareFetch } from "../../open-sse/utils/proxyFetch.js"; +import { getUsageForProvider } from "../../open-sse/services/usage.js"; +import { + USAGE_APIKEY_PROVIDERS, + USAGE_SUPPORTED_PROVIDERS, +} from "../../src/shared/constants/providers.js"; + +const USAGE_URL = "https://opencode.ai/zen/go/v1/usage"; + +function jsonResponse(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { "Content-Type": "application/json" }, + }); +} + +describe("OpenCode Go registry usage flags", () => { + it("is listed for the API key quota dashboard", () => { + expect(USAGE_SUPPORTED_PROVIDERS).toContain("opencode-go"); + expect(USAGE_APIKEY_PROVIDERS).toContain("opencode-go"); + }); +}); + +describe("getUsageForProvider(opencode-go)", () => { + beforeEach(() => vi.clearAllMocks()); + + it("fetches and normalizes subscription usage", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ + usage: { + rolling: { status: "ok", percent: 13, resetsAt: "2026-09-04T14:28:02.617Z" }, + weekly: { status: "ok", percent: 5, resetsAt: "2026-09-07T00:00:00.617Z" }, + monthly: { status: "ok", percent: 2, resetsAt: "2026-10-02T12:14:24.617Z" }, + }, + }), + ); + + const usage = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + + expect(proxyAwareFetch).toHaveBeenCalledWith( + USAGE_URL, + expect.objectContaining({ + method: "GET", + headers: expect.objectContaining({ Authorization: "Bearer sk-go-test" }), + }), + null, + ); + expect(usage).toEqual({ + plan: "OpenCode Go", + quotas: { + Rolling: { + used: 13, + total: 100, + remaining: 87, + remainingPercentage: 87, + resetAt: "2026-09-04T14:28:02.617Z", + unlimited: false, + }, + Weekly: expect.objectContaining({ used: 5, remainingPercentage: 95 }), + Monthly: expect.objectContaining({ used: 2, remainingPercentage: 98 }), + }, + }); + }); + + it("reports missing and rejected credentials", async () => { + const missing = await getUsageForProvider({ provider: "opencode-go" }); + expect(missing.message).toMatch(/api key/i); + expect(proxyAwareFetch).not.toHaveBeenCalled(); + + proxyAwareFetch.mockResolvedValueOnce(jsonResponse({ error: "unauthorized" }, 401)); + const rejected = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "bad", + }); + expect(rejected.message).toMatch(/authentication failed/i); + }); + + it("distinguishes a missing subscription from invalid credentials", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ error: { type: "EntitlementError" } }, 403), + ); + + const usage = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-without-go", + }); + + expect(usage.message).toMatch(/subscription required/i); + }); + + it("rejects responses without a valid quota percentage", async () => { + proxyAwareFetch.mockResolvedValueOnce( + jsonResponse({ usage: { rolling: { status: "ok" }, future: { percent: 10 } } }), + ); + + const usage = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + + expect(usage.quotas).toBeUndefined(); + expect(usage.message).toMatch(/valid quota data/i); + }); + + it("reports upstream and network failures", async () => { + proxyAwareFetch.mockResolvedValueOnce(jsonResponse({ error: "unavailable" }, 500)); + const upstream = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + expect(upstream.message).toContain("500"); + + proxyAwareFetch.mockRejectedValueOnce(new Error("socket closed")); + const network = await getUsageForProvider({ + provider: "opencode-go", + apiKey: "sk-go-test", + }); + expect(network.message).toContain("socket closed"); + }); +}); diff --git a/tests/unit/opencode-muse-spark-thinking.test.js b/tests/unit/opencode-muse-spark-thinking.test.js index 800f7d4e..94a44e8c 100644 --- a/tests/unit/opencode-muse-spark-thinking.test.js +++ b/tests/unit/opencode-muse-spark-thinking.test.js @@ -63,6 +63,39 @@ describe("OpenCode Free Muse Spark thinking", () => { expect(out.max_tokens).toBeUndefined(); }); + it("routes Union Alpha through Anthropic Messages", () => { + const caps = getCapabilitiesForModel(PROVIDER, "union-alpha"); + expect(caps.vision).toBe(true); + expect(caps.contextWindow).toBe(262144); + expect(caps.maxOutput).toBe(131072); + + const executor = new OpenCodeExecutor(); + + expect(getModelTargetFormat("oc", "union-alpha")).toBe(FORMATS.CLAUDE); + const url = executor.buildUrl("union-alpha"); + expect(url).toBe("https://opencode.ai/zen/v1/messages"); + expect(executor.buildHeaders({}, true, url)).toMatchObject({ + "anthropic-version": "2023-06-01", + }); + expect(executor.buildHeaders({}, true, executor.buildUrl("big-pickle"))) + .not.toHaveProperty("anthropic-version"); + + const translated = translateRequest( + FORMATS.OPENAI, + FORMATS.CLAUDE, + "union-alpha", + { messages: [{ role: "user", content: "ping" }], max_tokens: 1 }, + false, + {}, + PROVIDER, + ); + expect(translated).toMatchObject({ + model: "union-alpha", + messages: [{ role: "user", content: [{ type: "text", text: "ping" }] }], + max_tokens: 1, + }); + }); + it("leaves the other free models on Chat Completions", () => { const executor = new OpenCodeExecutor(); const body = { messages: [{ role: "user", content: "hi" }], max_tokens: 1024 }; @@ -132,4 +165,63 @@ describe("OpenCode Free Muse Spark thinking", () => { expect(out.max_tokens).toBeUndefined(); } }); + + it("strips prior-turn reasoning items carrying encrypted_content from input", () => { + const executor = new OpenCodeExecutor(); + const model = "muse-spark-1.3-contributor-free"; + const body = { + model, + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "say hi" }] }, + { + type: "reasoning", + id: "rs_123", + encrypted_content: "ENC_BLOB_TURN_1", + summary: [{ type: "summary_text", text: "thinking text" }], + }, + { + type: "function_call", + id: "fc_1", + call_id: "call_1", + name: "shell", + arguments: JSON.stringify({ command: "echo hi" }), + }, + { + type: "function_call_output", + call_id: "call_1", + output: "hi", + }, + { type: "message", role: "user", content: [{ type: "input_text", text: "now say bye" }] }, + ], + tools: [ + { + type: "function", + function: { + name: "shell", + description: "Run shell command", + parameters: { type: "object" }, + }, + }, + ], + }; + + const out = executor.transformRequest(model, body, true, {}); + expect(out.stream).toBe(true); + expect(out.store).toBe(false); + // Prior reasoning items stripped to prevent 400 "reasoning encrypted_content was not issued to this caller" + expect(out.input.some((item) => item.type === "reasoning")).toBe(false); + expect(JSON.stringify(out.input)).not.toContain("ENC_BLOB_TURN_1"); + // User message, function_call, function_call_output, and next user message survive + const types = out.input.map((item) => item.type); + expect(types).toEqual(["message", "function_call", "function_call_output", "message"]); + // Tools flattened and empty properties added + expect(out.tools).toEqual([ + { + type: "function", + name: "shell", + description: "Run shell command", + parameters: { type: "object", properties: {} }, + }, + ]); + }); }); diff --git a/tests/unit/opencode-session.test.js b/tests/unit/opencode-session.test.js new file mode 100644 index 00000000..3e50b9fe --- /dev/null +++ b/tests/unit/opencode-session.test.js @@ -0,0 +1,319 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; + +const { fetchMock } = vi.hoisted(() => ({ + fetchMock: vi.fn(), +})); + +vi.mock("../../open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: fetchMock, +})); + +import { getExecutor } from "../../open-sse/executors/index.js"; +import { + OPENCODE_SESSION_RE, + OPENCODE_REQUEST_RE, + generateSessionId, + generateRequestId, + translateSessionId, + stableSessionId, + deriveRequestId, +} from "../../open-sse/executors/opencode.js"; + +function makeCredentials(overrides = {}) { + return { + connectionId: "conn_test", + rawHeaders: {}, + ...overrides, + }; +} + +function prepare(executor, overrides = {}) { + const credentials = overrides.credentials || makeCredentials(); + const prepared = executor.prepareRequestCredentials({ + body: overrides.body || { input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }] }, + credentials, + providerSessionId: overrides.providerSessionId ?? "conversation-a", + clientTool: overrides.clientTool ?? "claude", + }); + return { credentials, prepared }; +} + +beforeEach(() => { + fetchMock.mockReset(); + fetchMock.mockResolvedValue(new Response("{}", { + status: 200, + headers: { "content-type": "application/json" }, + })); +}); + +describe("OpenCode Free Session ID Format", () => { + it("generates session IDs matching OpenCode canonical format (ses_ + 12 hex + 14 base62)", () => { + for (let i = 0; i < 20; i++) { + const id = generateSessionId(); + expect(id).toMatch(OPENCODE_SESSION_RE); + expect(id).toHaveLength(30); + } + }); + + it("generates request IDs matching OpenCode canonical format (msg_ + 12 hex + 14 base62)", () => { + for (let i = 0; i < 20; i++) { + const id = generateRequestId(); + expect(id).toMatch(/^msg_[0-9a-f]{12}[0-9A-Za-z]{14}$/); + expect(id).toHaveLength(30); + } + }); + + it("translates arbitrary sessions into valid OpenCode session format", () => { + const inputs = [ + "claude:550e8400-e29b-41d4-a716-446655440000", + "antigravity:conv-abc-123", + "session-from-codex", + "12345", + "", + ]; + for (const raw of inputs) { + const translated = translateSessionId(raw, "claude"); + expect(translated).toMatch(OPENCODE_SESSION_RE); + expect(translated).toHaveLength(30); + } + }); + + it("preserves already-valid OpenCode sessions without re-hashing", () => { + const valid = "ses_f534dfae8ffeCy4Ee4tLWNygDc"; + expect(translateSessionId(valid)).toBe(valid); + expect(translateSessionId(` ${valid} `)).toBe(valid); + }); +}); + +describe("OpenCode Free Executor Session Resolution", () => { + it("uses request-local session credentials without mutating source credentials", () => { + const executor = getExecutor("opencode"); + const { credentials, prepared } = prepare(executor); + + expect(executor.constructor.name).toBe("OpenCodeExecutor"); + expect(prepared).not.toBe(credentials); + expect(prepared._opencodeSession).toMatch(OPENCODE_SESSION_RE); + expect(credentials).not.toHaveProperty("_opencodeSession"); + expect(executor).not.toHaveProperty("_currentSessionId"); + }); + + it("preserves valid native x-opencode-session header case-insensitively", () => { + const executor = getExecutor("opencode"); + const valid = "ses_f534dfae8ffeCy4Ee4tLWNygDc"; + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "X-OpenCode-Session": ` ${valid} ` } }), + }); + + expect(prepared._opencodeSession).toBe(valid); + }); + + it("translates invalid native x-opencode-session header into a valid session", () => { + const executor = getExecutor("opencode"); + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "x-opencode-session": "invalid-session-uuid" } }), + }); + + expect(prepared._opencodeSession).toMatch(OPENCODE_SESSION_RE); + expect(prepared._opencodeSession).not.toBe("invalid-session-uuid"); + }); + + it("translates conversation session deterministically", () => { + const executor = getExecutor("opencode"); + const first = prepare(executor, { providerSessionId: "conversation-a", clientTool: "claude" }).prepared._opencodeSession; + const second = prepare(executor, { providerSessionId: "conversation-a", clientTool: "claude" }).prepared._opencodeSession; + + expect(first).toBe(second); + expect(first).toMatch(OPENCODE_SESSION_RE); + }); + + it("isolates different conversations and tools", () => { + const executor = getExecutor("opencode"); + const convA = prepare(executor, { providerSessionId: "conversation-a" }).prepared._opencodeSession; + const convB = prepare(executor, { providerSessionId: "conversation-b" }).prepared._opencodeSession; + const toolClaude = prepare(executor, { providerSessionId: "same", clientTool: "claude" }).prepared._opencodeSession; + const toolCodex = prepare(executor, { providerSessionId: "same", clientTool: "codex" }).prepared._opencodeSession; + + expect(convA).not.toBe(convB); + expect(toolClaude).not.toBe(toolCodex); + }); + + it("adds the valid session header to fetch requests", async () => { + const executor = getExecutor("opencode"); + const credentials = makeCredentials(); + const result = await executor.execute({ + model: "muse-spark-1.3-contributor-free", + body: { input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hello" }] }] }, + stream: false, + credentials, + providerSessionId: "conversation-fetch-test", + clientTool: "claude", + }); + + expect(result.headers["x-opencode-session"]).toMatch(OPENCODE_SESSION_RE); + expect(fetchMock).toHaveBeenCalledOnce(); + expect(fetchMock.mock.calls[0][1].headers["x-opencode-session"]).toBe(result.headers["x-opencode-session"]); + expect(fetchMock.mock.calls[0][1].headers["Authorization"]).toBe("Bearer public"); + expect(credentials).not.toHaveProperty("_opencodeSession"); + }); + + it("falls back to a valid generated session in buildHeaders when called standalone", () => { + const executor = getExecutor("opencode"); + const headers = executor.buildHeaders({}); + + expect(headers["x-opencode-session"]).toMatch(OPENCODE_SESSION_RE); + expect(headers["Authorization"]).toBe("Bearer public"); + }); + it("handles null or undefined body gracefully in transformRequest", () => { + const executor = getExecutor("opencode"); + expect(() => executor.transformRequest("muse-spark-1.3-contributor-free", null, false, {})).not.toThrow(); + expect(() => executor.transformRequest("big-pickle", undefined, false, {})).not.toThrow(); + }); +}); + +describe("OpenCode Free User-Agent Validation", () => { + it("defaults User-Agent to opencode/1.18.31 for non-opencode downstream clients", () => { + const executor = getExecutor("opencode"); + const headersNoUa = executor.buildHeaders({}); + expect(headersNoUa["User-Agent"]).toBe("opencode/1.18.31"); + + const headersClaude = executor.buildHeaders({ rawHeaders: { "user-agent": "Claude-Code/1.0" } }); + expect(headersClaude["User-Agent"]).toBe("opencode/1.18.31"); + }); + + it("replaces bare opencode with versioned opencode/1.18.31 to prevent 403 FreeTierError", () => { + const executor = getExecutor("opencode"); + const headers = executor.buildHeaders({ rawHeaders: { "user-agent": "opencode" } }); + expect(headers["User-Agent"]).toBe("opencode/1.18.31"); + }); + + it("upgrades outdated opencode versions (< 1.17) to prevent 426 Upgrade Required", () => { + const executor = getExecutor("opencode"); + const headers = executor.buildHeaders({ rawHeaders: { "user-agent": "opencode/1.15.0" } }); + expect(headers["User-Agent"]).toBe("opencode/1.18.31"); + }); + + it("preserves valid opencode versions (>= 1.17)", () => { + const executor = getExecutor("opencode"); + const headers118 = executor.buildHeaders({ + rawHeaders: { "user-agent": "opencode/1.18.31 ai-sdk/provider-utils/4.0.40 runtime/bun/1.3.14" }, + }); + expect(headers118["User-Agent"]).toBe("opencode/1.18.31 ai-sdk/provider-utils/4.0.40 runtime/bun/1.3.14"); + + const headersFuture = executor.buildHeaders({ rawHeaders: { "user-agent": "opencode/1.19.0" } }); + expect(headersFuture["User-Agent"]).toBe("opencode/1.19.0"); + }); +}); + +describe("OpenCode Stable Session Reuse (429 follow-up)", () => { + function anonymousCredentials(auth) { + return makeCredentials({ connectionId: undefined, rawHeaders: { authorization: `Bearer ${auth}` } }); + } + + it("reuses one stable upstream session instead of minting a new one per request", () => { + const executor = getExecutor("opencode"); + const body = { messages: [{ role: "user", content: "hello" }] }; + const first = executor.prepareRequestCredentials({ + body, + credentials: anonymousCredentials("stable-key-1"), + providerSessionId: null, + clientTool: "claude", + }); + const second = executor.prepareRequestCredentials({ + body, + credentials: anonymousCredentials("stable-key-1"), + providerSessionId: null, + clientTool: "claude", + }); + + expect(first._opencodeSession).toMatch(OPENCODE_SESSION_RE); + expect(second._opencodeSession).toBe(first._opencodeSession); + }); + + it("isolates stable sessions by downstream identity", () => { + const executor = getExecutor("opencode"); + const body = { messages: [{ role: "user", content: "hello" }] }; + const forKey = (auth) => executor.prepareRequestCredentials({ + body, + credentials: anonymousCredentials(auth), + providerSessionId: null, + clientTool: "claude", + })._opencodeSession; + + expect(forKey("user-A")).not.toBe(forKey("user-B")); + expect(forKey("user-A")).toMatch(OPENCODE_SESSION_RE); + }); + + it("exposes the stable session helper directly", () => { + const first = stableSessionId({ connectionId: "direct-conn" }); + expect(stableSessionId({ connectionId: "direct-conn" })).toBe(first); + expect(first).toMatch(OPENCODE_SESSION_RE); + }); + + it("derives deterministic, canonical request ids per message", () => { + const session = stableSessionId({ connectionId: "req-conn" }); + const body = { messages: [{ role: "user", content: "ping" }] }; + const first = deriveRequestId(session, body); + expect(first).toMatch(OPENCODE_REQUEST_RE); + expect(deriveRequestId(session, body)).toBe(first); + expect( + deriveRequestId(session, { messages: [{ role: "user", content: "a different question" }] }), + ).not.toBe(first); + }); + + it("preserves a valid downstream x-opencode-request header", () => { + const executor = getExecutor("opencode"); + const validReq = "msg_0ae8d9cd3001swxaFbM248jcIF"; + const { prepared } = prepare(executor, { + credentials: makeCredentials({ rawHeaders: { "x-opencode-request": validReq } }), + }); + expect(prepared._opencodeRequest).toBe(validReq); + }); + + it("keeps the standalone buildHeaders session stable across calls", () => { + const executor = getExecutor("opencode"); + const first = executor.buildHeaders({})["x-opencode-session"]; + const second = executor.buildHeaders({})["x-opencode-session"]; + expect(first).toMatch(OPENCODE_SESSION_RE); + expect(second).toBe(first); + }); + + it("cloaks free-tier requests with bash and read decoy tools", () => { + const executor = getExecutor("opencode"); + + // Case 1: no tools sent by client -> injects bash + read with tool_choice none + const chatNoTools = executor.transformRequest("nemotron-3-ultra-free", { + messages: [{ role: "user", content: "hi" }], + }); + expect(chatNoTools.stream).toBe(true); + expect(chatNoTools.tool_choice).toBe("none"); + expect(chatNoTools.tools.map((t) => t.function?.name)).toEqual(["bash", "read"]); + + // Case 2: external CLI tools (e.g. Claude Code Bash) -> preserves Bash, appends read + const chatWithTools = executor.transformRequest("nemotron-3-ultra-free", { + messages: [{ role: "user", content: "hi" }], + tools: [{ type: "function", function: { name: "Bash", description: "Claude Code tool" } }], + tool_choice: "auto", + }); + expect(chatWithTools.tool_choice).toBe("auto"); + const names = chatWithTools.tools.map((t) => t.function?.name); + expect(names).toContain("Bash"); + expect(names).toContain("bash"); + expect(names).toContain("read"); + + // Case 3: already has both bash and read -> do not insert anything + const chatFull = executor.transformRequest("nemotron-3-ultra-free", { + messages: [{ role: "user", content: "hi" }], + tools: [ + { type: "function", function: { name: "bash", description: "existing" } }, + { type: "function", function: { name: "read", description: "existing" } }, + ], + }); + expect(chatFull.tools.length).toBe(2); + expect(chatFull.tools[0].function.description).toBe("existing"); + }); + + it("declares forceStream on the opencode transport so chatCore serves SSE upstream", async () => { + const { PROVIDERS } = await import("../../open-sse/config/providers.js"); + expect(PROVIDERS.opencode?.forceStream).toBe(true); + }); +}); diff --git a/tests/unit/prefetch-images.test.js b/tests/unit/prefetch-images.test.js index 2289697b..0d4c0db3 100644 --- a/tests/unit/prefetch-images.test.js +++ b/tests/unit/prefetch-images.test.js @@ -55,4 +55,21 @@ describe("prefetchRemoteImages", () => { expect(n).toBe(1); expect(body.messages[0].content[0].source.type).toBe("base64"); }); + + it("openai source -> commandcode target: converts remote URL to base64", async () => { + const body = { messages: [{ role: "user", content: [{ type: "image_url", image_url: { url: "https://x/a.png" } }] }] }; + const n = await prefetchRemoteImages(body, FORMATS.OPENAI, FORMATS.COMMANDCODE); + expect(n).toBe(1); + expect(body.messages[0].content[0].image_url.url.startsWith("data:image/png;base64,")).toBe(true); + expect(fetchImageAsBase64).toHaveBeenCalled(); + }); + + it("claude source -> commandcode target: source.url -> base64", async () => { + const body = { messages: [{ role: "user", content: [ + { type: "image", source: { type: "url", url: "https://x/a.png" } }, + ] }] }; + const n = await prefetchRemoteImages(body, FORMATS.CLAUDE, FORMATS.COMMANDCODE); + expect(n).toBe(1); + expect(body.messages[0].content[0].source.type).toBe("base64"); + }); }); diff --git a/tests/unit/provider-quota-visibility.test.js b/tests/unit/provider-quota-visibility.test.js index d6616b81..8c69ee62 100644 --- a/tests/unit/provider-quota-visibility.test.js +++ b/tests/unit/provider-quota-visibility.test.js @@ -3,6 +3,7 @@ import { filterQuotasByVisibility, getHiddenQuotaRows, parseQuotaData, + trimHiddenQuotaKeys, } from "@/app/(dashboard)/dashboard/usage/components/ProviderLimits/utils.js"; describe("provider quota visibility", () => { @@ -13,22 +14,26 @@ describe("provider quota visibility", () => { used: 200, total: 1000, resetAt: "2026-07-04T00:00:00Z", + remainingPercentage: 80, }, "claude-opus-4-6-thinking": { displayName: "Claude Opus 4.6 (Thinking)", used: 100, total: 1000, resetAt: "2026-07-04T00:00:00Z", + remainingPercentage: 90, }, }, }; - it("keeps Antigravity modelKey so hidden settings use stable quota ids", () => { + it("groups Antigravity model quotas into Gemini and Claude families", () => { const quotas = parseQuotaData("antigravity", data); expect(quotas.map((q) => q.modelKey)).toEqual([ - "gemini-pro-agent", - "claude-opus-4-6-thinking", + "gemini", + "claude", ]); + expect(quotas[0].name).toBe("Gemini (Flash / Pro)"); + expect(quotas[1].name).toBe("Claude (Sonnet / Opus)"); }); it("shows all quotas by default and hides configured provider rows", () => { @@ -36,19 +41,34 @@ describe("provider quota visibility", () => { expect(filterQuotasByVisibility("antigravity", quotas, {})).toHaveLength(2); const visibility = { - antigravity: { hidden: ["claude-opus-4-6-thinking"] }, + antigravity: { hidden: ["claude"] }, }; const visible = filterQuotasByVisibility("antigravity", quotas, visibility); const hidden = getHiddenQuotaRows("antigravity", quotas, visibility); - expect(visible.map((q) => q.modelKey)).toEqual(["gemini-pro-agent"]); - expect(hidden.map((q) => q.modelKey)).toEqual(["claude-opus-4-6-thinking"]); + expect(visible.map((q) => q.modelKey)).toEqual(["gemini"]); + expect(hidden.map((q) => q.modelKey)).toEqual(["claude"]); + }); + + it("trims stale or obsolete model keys", () => { + const quotas = parseQuotaData("antigravity", data); + const trimmed = trimHiddenQuotaKeys(["claude", "stale-model-xyz", "gemini-3.8-flash-low"], quotas); + expect(trimmed).toEqual(["claude"]); + + const visibility = { + antigravity: { hidden: ["claude", "stale-model-xyz"] }, + }; + const visible = filterQuotasByVisibility("antigravity", quotas, visibility); + const hidden = getHiddenQuotaRows("antigravity", quotas, visibility); + + expect(visible.map((q) => q.modelKey)).toEqual(["gemini"]); + expect(hidden.map((q) => q.modelKey)).toEqual(["claude"]); }); it("does not apply one provider hidden list to another provider", () => { const quotas = parseQuotaData("antigravity", data); const visibility = { - codex: { hidden: ["gemini-pro-agent"] }, + codex: { hidden: ["gemini"] }, }; expect(filterQuotasByVisibility("antigravity", quotas, visibility)).toHaveLength(2); }); diff --git a/tests/unit/qoder-context-tier.test.js b/tests/unit/qoder-context-tier.test.js new file mode 100644 index 00000000..8850fd0d --- /dev/null +++ b/tests/unit/qoder-context-tier.test.js @@ -0,0 +1,193 @@ +/** + * Qoder context-window tiers + routable model listing. + * + * The Qoder IDE lets a user pick 200K / 400K / 1M for a model; qodercli-style + * requests (what 9router sends) only carry the default max_input_tokens. These + * tests pin the escalation policy and the payload fields the IDE writes. + */ +import { describe, it, expect } from "vitest"; + +import { + parseTierTokenCount, + getQoderContextTiers, + estimateQoderPromptTokens, + resolveQoderContextTier, + applyQoderContextTier, +} from "../../open-sse/shared/qoder/contextTier.js"; +import { routableQoderModels } from "../../open-sse/services/qoderModels.js"; + +// Shape mirrors the live /algo/api/v2/model/list entry for qmodel_38max. +const MODEL_CONFIG = { + key: "qmodel_38max", + display_name: "Qwen3.8-Max", + is_reasoning: true, + max_input_tokens: 180_000, + max_output_tokens: 32_768, + context_config: [ + { name: "200K", tokenCount: 200_000, isDefault: true }, + { name: "400K", tokenCount: 400_000, isDefault: false }, + { name: "1M", tokenCount: 1_000_000, isDefault: false }, + ], +}; + +function promptOfTokens(n) { + // ~4 ASCII chars per token + return { system: "", messages: [{ role: "user", content: "abcd".repeat(n) }], tools: [] }; +} + +describe("parseTierTokenCount", () => { + it("accepts numbers and K/M suffixed strings", () => { + expect(parseTierTokenCount(204800)).toBe(204800); + expect(parseTierTokenCount("200K")).toBe(200_000); + expect(parseTierTokenCount("1M")).toBe(1_000_000); + expect(parseTierTokenCount("1.5m")).toBe(1_500_000); + expect(parseTierTokenCount("131072")).toBe(131072); + }); + + it("returns 0 for garbage", () => { + expect(parseTierTokenCount(null)).toBe(0); + expect(parseTierTokenCount("big")).toBe(0); + expect(parseTierTokenCount(-5)).toBe(0); + }); +}); + +describe("getQoderContextTiers", () => { + it("sorts tiers ascending and keeps the default flag", () => { + const tiers = getQoderContextTiers({ + context_config: [ + { name: "1M", tokenCount: 1_000_000 }, + { name: "200K", tokenCount: 200_000, isDefault: true }, + ], + }); + expect(tiers.map((t) => t.tokenCount)).toEqual([200_000, 1_000_000]); + expect(tiers[0].isDefault).toBe(true); + expect(tiers[1].isDefault).toBe(false); + }); + + it("understands camelCase / snake_case variants and derives names", () => { + const tiers = getQoderContextTiers({ + contextConfig: [{ token_count: "400K", is_default: true }, { max_input_tokens: 1_000_000 }], + }); + expect(tiers).toEqual([ + { name: "400K", tokenCount: 400_000, isDefault: true }, + { name: "1M", tokenCount: 1_000_000, isDefault: false }, + ]); + }); + + it("returns [] when the model has no tiers", () => { + expect(getQoderContextTiers({ max_input_tokens: 131072 })).toEqual([]); + expect(getQoderContextTiers(null)).toEqual([]); + }); +}); + +describe("estimateQoderPromptTokens", () => { + it("counts CJK characters as ~1 token each instead of chars/4", () => { + const ascii = estimateQoderPromptTokens({ messages: [{ role: "user", content: "a".repeat(4000) }] }); + const cjk = estimateQoderPromptTokens({ messages: [{ role: "user", content: "中".repeat(4000) }] }); + expect(ascii).toBeLessThan(1_200); + expect(cjk).toBeGreaterThan(4_000); + }); +}); + +describe("resolveQoderContextTier (auto)", () => { + it("leaves the payload untouched while the prompt fits the current max_input_tokens", () => { + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(50_000))).toBeNull(); + }); + + it("escalates to the smallest tier that fits once the prompt outgrows the default", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(250_000)); + expect(choice).not.toBeNull(); + expect(choice.tier.name).toBe("400K"); + expect(choice.reason).toBe("auto:fits"); + expect(choice.estimatedTokens).toBeGreaterThan(240_000); + }); + + it("falls back to the largest tier when nothing fits (upstream decides)", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(1_200_000)); + expect(choice.tier.name).toBe("1M"); + expect(choice.reason).toBe("auto:largest"); + }); + + it("applies headroom so a prompt just under the limit still escalates", () => { + // 170K estimated * 1.15 = 195.5K > 180K current → smallest tier above the current limit (200K) + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(170_000))?.tier.name).toBe("200K"); + // 190K * 1.15 = 218.5K → 200K no longer fits → 400K + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(190_000))?.tier.name).toBe("400K"); + }); + + it("returns null for models without context_config", () => { + expect(resolveQoderContextTier({ max_input_tokens: 131072 }, promptOfTokens(500_000))).toBeNull(); + }); + + it("never escalates when the current limit is already the largest tier", () => { + const cfg = { ...MODEL_CONFIG, max_input_tokens: 1_000_000 }; + expect(resolveQoderContextTier(cfg, promptOfTokens(1_500_000))).toBeNull(); + }); +}); + +describe("resolveQoderContextTier (forced via QODER_CONTEXT_TIER)", () => { + it("max picks the largest tier regardless of prompt size", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "max" }); + expect(choice.tier.name).toBe("1M"); + expect(choice.reason).toBe("forced:max"); + }); + + it("default picks the isDefault tier", () => { + const choice = resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "default" }); + expect(choice.tier.name).toBe("200K"); + }); + + it("a tier name or token count selects that tier", () => { + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "400k" }).tier.tokenCount).toBe(400_000); + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "1000000" }).tier.name).toBe("1M"); + }); + + it("an unknown tier name falls back to auto", () => { + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(10), { preference: "9M" })).toBeNull(); + expect(resolveQoderContextTier(MODEL_CONFIG, promptOfTokens(250_000), { preference: "9M" }).tier.name).toBe("400K"); + }); +}); + +describe("applyQoderContextTier", () => { + it("mirrors the tier into the three places the IDE writes", () => { + const payload = { + parameters: { max_tokens: 32_768 }, + chat_context: { extra: { context: [], modelConfig: { key: "qmodel_38max" } } }, + model_config: { ...MODEL_CONFIG }, + }; + applyQoderContextTier(payload, { name: "1M", tokenCount: 1_000_000 }); + expect(payload.parameters).toEqual({ max_tokens: 32_768, context_length: 1_000_000 }); + expect(payload.chat_context.extra.ideModelConfigOverride).toEqual({ max_input_tokens: 1_000_000 }); + expect(payload.chat_context.extra.modelConfig).toEqual({ key: "qmodel_38max" }); + expect(payload.model_config.max_input_tokens).toBe(1_000_000); + expect(payload.model_config.context_config).toHaveLength(3); + }); + + it("is a no-op without a tier", () => { + const payload = { parameters: { max_tokens: 1 } }; + expect(applyQoderContextTier(payload, null)).toBe(payload); + expect(payload).toEqual({ parameters: { max_tokens: 1 } }); + }); +}); + +describe("routableQoderModels", () => { + it("lists visible models first, then hidden (enable:false) catalog keys", () => { + const catalog = { + models: [{ id: "qmodel_38max", name: "Qwen3.8-Max" }], + rawConfigs: new Map([ + ["qmodel_38max", { key: "qmodel_38max", enable: true }], + ["qfmodel", { key: "qfmodel", enable: false, display_name: "Qwen Fast" }], + ["dmodel", { key: "dmodel", enable: false }], + ]), + }; + expect(routableQoderModels(catalog)).toEqual([ + { id: "qmodel_38max", name: "Qwen3.8-Max", hidden: false }, + { id: "qfmodel", name: "Qwen Fast", hidden: true }, + { id: "dmodel", name: "dmodel", hidden: true }, + ]); + }); + + it("returns [] for a failed catalog fetch", () => { + expect(routableQoderModels(null)).toEqual([]); + }); +}); diff --git a/tests/unit/qoder.test.js b/tests/unit/qoder.test.js index d8fce1ae..ea45ac61 100644 --- a/tests/unit/qoder.test.js +++ b/tests/unit/qoder.test.js @@ -9,7 +9,7 @@ * - device flow URL construction */ -import { describe, it, expect } from "vitest"; +import { describe, it, expect, beforeEach } from "vitest"; import crypto from "crypto"; import { qoderEncodeBody } from "../../src/lib/qoder/encoding.js"; @@ -22,6 +22,13 @@ import { } from "../../src/lib/qoder/constants.js"; import { PROVIDER_MODELS } from "../../open-sse/config/providerModels.js"; import { __test__ as qoderExecutorInternals } from "../../open-sse/executors/qoder.js"; +import { canonicalizeQoderUsage } from "../../open-sse/shared/qoder/sse.js"; +import { + rewriteQoderMessageAttachments, + clearQoderUploadCache, + buildMultipartFile, +} from "../../open-sse/shared/qoder/attachments.js"; +import { qoderInferenceBase } from "../../open-sse/shared/qoder/constants.js"; // Convenience aliases — tests were originally written against module-level // helpers; the QoderService class wraps them so each test creates its own @@ -367,6 +374,85 @@ describe("normalizeMessages", () => { expect(result.messages).toEqual([]); expect(result.systemText).toBe(""); }); + + it("preserves image_url blocks (http URL) instead of dropping them", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "describe" }, + { type: "image_url", image_url: { url: "https://example.com/a.png" } }, + ], + }, + ]); + const content = result.messages[0].content; + expect(Array.isArray(content)).toBe(true); + expect(content).toContainEqual({ type: "text", text: "describe" }); + expect(content).toContainEqual({ type: "image_url", image_url: { url: "https://example.com/a.png" } }); + }); + + it("preserves base64 data: URI images (no OSS upload needed)", () => { + const dataUri = "data:image/png;base64,iVBORw0KGgo="; + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "image_url", image_url: { url: dataUri } }, + { type: "text", text: "what color?" }, + ], + }, + ]); + const content = result.messages[0].content; + expect(Array.isArray(content)).toBe(true); + expect(content[0]).toEqual({ type: "image_url", image_url: { url: dataUri } }); + expect(content.some((b) => b.type === "text" && b.text === "what color?")).toBe(true); + }); + + it("converts claude-style base64 image blocks to image_url data URIs", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "see this" }, + { type: "image", source: { type: "base64", media_type: "image/jpeg", data: "AAAA" } }, + ], + }, + ]); + const content = result.messages[0].content; + expect(content).toContainEqual({ + type: "image_url", + image_url: { url: "data:image/jpeg;base64,AAAA" }, + }); + }); + + it("drops image blocks with no usable url but keeps the text", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "hi" }, + { type: "image_url", image_url: {} }, + { type: "image", source: { type: "base64" } }, + ], + }, + ]); + expect(result.messages[0].content).toBe("hi"); + }); + + it("turns leftover file/document blocks into short stubs instead of dropping them", () => { + const result = normalizeMessages([ + { + role: "user", + content: [ + { type: "text", text: "see" }, + { type: "file", file: { filename: "big.pdf", file_data: "data:application/pdf;base64,AAA" } }, + ], + }, + ]); + expect(result.messages[0].content).toContain("see"); + expect(result.messages[0].content).toContain("big.pdf"); + expect(result.messages[0].content).not.toContain("AAA"); + }); }); describe("wrapQoderSSE", () => { @@ -466,4 +552,190 @@ describe("wrapQoderSSE", () => { const wrapped = await wrapQoderSSE(r, "qoder/auto"); expect(wrapped).toBe(r); }); + + function envelope(body) { + return `data: ${JSON.stringify({ statusCodeValue: 200, body })}\n\n`; + } + + function parseForwardedChunks(out) { + return out + .split("\n\n") + .map((block) => block.trim()) + .filter((block) => block.startsWith("data:") && !block.includes("[DONE]")) + .map((block) => JSON.parse(block.slice("data:".length).trim())); + } + + it("coalesces empty finish-in-delta + usage-only into one OpenAI usage chunk", async () => { + const content = JSON.stringify({ + id: "chatcmpl-qoder-1", + created: 1700000000, + model: "auto", + choices: [{ index: 0, delta: { content: "hi" } }], + }); + const finish = JSON.stringify({ + id: "chatcmpl-qoder-1", + choices: [{ index: 0, delta: { content: "", finish_reason: "stop" } }], + }); + const usage = JSON.stringify({ + id: "chatcmpl-qoder-1", + choices: [], + usage: { + prompt_tokens: 100, + completion_tokens: 20, + total_tokens: 120, + prompt_tokens_details: { cached_tokens: 40 }, + }, + }); + const wrapped = await wrapQoderSSE( + makeResponse([envelope(content) + envelope(finish) + envelope(usage) + envelope("[DONE]")]), + "qoder/auto", + ); + const out = await drain(wrapped); + expect(out).toContain(`data: ${content}\n\n`); + const chunks = parseForwardedChunks(out); + const usageChunk = chunks.find((c) => c.usage); + expect(usageChunk).toBeDefined(); + expect(usageChunk.choices[0].finish_reason).toBe("stop"); + expect(usageChunk.usage.prompt_tokens).toBe(100); + expect(usageChunk.usage.completion_tokens).toBe(20); + expect(usageChunk.usage.prompt_tokens_details.cached_tokens).toBe(40); + expect(chunks.some((c) => Array.isArray(c.choices) && c.choices.length === 0)).toBe(false); + expect((out.match(/data: \[DONE\]/g) || []).length).toBe(1); + }); + + it("maps Qoder input_tokens aliases onto prompt_tokens in the coalesced usage chunk", async () => { + const finish = JSON.stringify({ + choices: [{ index: 0, delta: { finish_reason: "stop" } }], + }); + const usage = JSON.stringify({ + choices: [], + usage: { + input_tokens: 80, + output_tokens: 10, + cache_read_input_tokens: 25, + }, + }); + const wrapped = await wrapQoderSSE( + makeResponse([envelope(finish) + envelope(usage)]), + "qoder/lite", + ); + const chunks = parseForwardedChunks(await drain(wrapped)); + const usageChunk = chunks.find((c) => c.usage); + expect(usageChunk.usage.prompt_tokens).toBe(80); + expect(usageChunk.usage.completion_tokens).toBe(10); + expect(usageChunk.usage.prompt_tokens_details.cached_tokens).toBe(25); + }); +}); + +describe("canonicalizeQoderUsage", () => { + it("returns null for missing or empty usage", () => { + expect(canonicalizeQoderUsage(null)).toBeNull(); + expect(canonicalizeQoderUsage({})).toBeNull(); + }); + + it("copies prompt_tokens_details.cached_tokens through", () => { + const out = canonicalizeQoderUsage({ + prompt_tokens: 50, + completion_tokens: 5, + prompt_tokens_details: { cached_tokens: 12 }, + }); + expect(out.prompt_tokens).toBe(50); + expect(out.cached_tokens).toBe(12); + expect(out.prompt_tokens_details.cached_tokens).toBe(12); + expect(out.total_tokens).toBe(55); + }); +}); + +describe("qoderInferenceBase", () => { + it("sends job tokens to api2 and device tokens to api3", () => { + expect(qoderInferenceBase({ accessToken: "jt-abc" })).toContain("api2.qoder.sh"); + expect(qoderInferenceBase({ accessToken: "dt-abc" })).toContain("api3.qoder.sh"); + }); +}); + +describe("rewriteQoderMessageAttachments", () => { + beforeEach(() => clearQoderUploadCache()); + + it("uploads data-URI images and keeps only the OSS URL in the message", async () => { + const messages = [{ + role: "user", + content: [ + { type: "text", text: "see this" }, + { type: "image_url", image_url: { url: "data:image/png;base64,AAAA" } }, + ], + }]; + const stats = await rewriteQoderMessageAttachments(messages, { + uploadFn: async ({ buffer, mediaType }) => { + expect(Buffer.isBuffer(buffer)).toBe(true); + expect(mediaType).toBe("image/png"); + return "https://cdn.qoder.example/img.png"; + }, + }); + expect(messages[0].content).toEqual([ + { type: "text", text: "see this" }, + { type: "image_url", image_url: { url: "https://cdn.qoder.example/img.png" } }, + ]); + expect(JSON.stringify(messages)).not.toContain("AAAA"); + expect(stats.imageUrls).toEqual(["https://cdn.qoder.example/img.png"]); + }); + + it("does not re-upload already-hosted http(s) image URLs", async () => { + const messages = [{ + role: "user", + content: [{ type: "image_url", image_url: { url: "https://example.com/a.png" } }], + }]; + await rewriteQoderMessageAttachments(messages, { + uploadFn: async () => { + throw new Error("should not upload remote URLs"); + }, + }); + expect(messages[0].content[0].image_url.url).toBe("https://example.com/a.png"); + }); + + it("stubs non-image file blocks instead of inlining bytes", async () => { + const pdfB64 = "A".repeat(200); + const messages = [{ + role: "user", + content: [ + { type: "text", text: "read this" }, + { type: "file", file: { filename: "big.pdf", file_data: `data:application/pdf;base64,${pdfB64}` } }, + ], + }]; + await rewriteQoderMessageAttachments(messages, { + uploadFn: async () => { + throw new Error("should not upload PDFs as images"); + }, + }); + const wire = JSON.stringify(messages); + expect(wire).not.toContain(pdfB64); + expect(wire).toContain("[file omitted: big.pdf"); + }); + + it("stubs oversized images when OSS upload fails instead of keeping a huge data URI", async () => { + const big = "A".repeat(700_000); + const messages = [{ + role: "user", + content: [{ type: "image_url", image_url: { url: `data:image/png;base64,${big}` } }], + }]; + await rewriteQoderMessageAttachments(messages, { + uploadFn: async () => { + throw new Error("upstream 413"); + }, + }); + const wire = JSON.stringify(messages); + expect(wire).not.toContain(big); + expect(wire).toContain("[file omitted:"); + expect(Buffer.byteLength(wire, "utf8")).toBeLessThan(4096); + }); + + it("buildMultipartFile uses the file field name qodercli sends", () => { + const { boundary, body } = buildMultipartFile(Buffer.from("hi"), { + fileName: "image.png", + mediaType: "image/png", + }); + const text = body.toString("latin1"); + expect(text).toContain(`name="file"`); + expect(text).toContain("filename=\"image.png\""); + expect(text).toContain(`--${boundary}`); + }); }); diff --git a/tests/unit/responses-abort-terminal.test.js b/tests/unit/responses-abort-terminal.test.js index 10af5848..e05c9d14 100644 --- a/tests/unit/responses-abort-terminal.test.js +++ b/tests/unit/responses-abort-terminal.test.js @@ -1,7 +1,9 @@ import { describe, expect, it } from "vitest"; -import { createDisconnectAwareStream } from "../../open-sse/utils/streamHandler.js"; +import { createDisconnectAwareStream, pipeWithDisconnect, createStreamController } from "../../open-sse/utils/streamHandler.js"; import { buildAbortedResponsesTerminalBytes } from "../../open-sse/utils/responsesStreamHelpers.js"; +import { buildStreamErrorBytes } from "../../open-sse/utils/streamHelpers.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; // Minimal stream controller stub function makeController() { @@ -70,3 +72,71 @@ describe("Responses abort terminal synthesis", () => { expect(text).not.toContain("[DONE]"); }); }); + +// A stream that aborts after HTTP 200 cannot change status, so the failure must +// travel in-band: structured error frame first, then [DONE]. openai-python raises +// APIError on any `data:` payload carrying an `error` key (checked before [DONE]); +// Anthropic clients need `event: error`. Never a fabricated finish_reason. +describe("buildStreamErrorBytes", () => { + const jsonOf = (sse) => JSON.parse(sse.match(/\{.*\}/s)[0]); + const textOf = (bytes) => new TextDecoder().decode(bytes); + + // onAbortTerminal callbacks are enqueued verbatim, so a string here is a + // silent no-op at runtime (createDisconnectAwareStream swallows the throw). + it("returns bytes, not a string", () => { + expect(buildStreamErrorBytes(504, "x", FORMATS.OPENAI)).toBeInstanceOf(Uint8Array); + }); + + it("emits error frame then [DONE] for OpenAI clients", () => { + const out = textOf(buildStreamErrorBytes(504, "stream stall timeout", FORMATS.OPENAI)); + + expect(out).toContain('data: {"error"'); + expect(out.indexOf("data: [DONE]")).toBeGreaterThan(out.indexOf('data: {"error"')); + + expect(jsonOf(out).error).toEqual({ + message: "stream stall timeout", + type: "server_error", + code: "gateway_timeout", + }); + }); + + it("emits event: error (no [DONE]) for Claude clients", () => { + const out = textOf(buildStreamErrorBytes(504, "stream stall timeout", FORMATS.CLAUDE)); + + expect(out).toContain("event: error\n"); + expect(out).not.toContain("[DONE]"); + expect(jsonOf(out)).toMatchObject({ type: "error", error: { message: "stream stall timeout" } }); + }); +}); + +// The wiring, not just the frame builder: the watchdog must hand its reason to +// onAbortTerminal and the bytes must reach a real consumer. +describe("stall abort through pipeWithDisconnect", () => { + it("delivers the error frame and closes the stream", async () => { + // Real controller: the stub above never fires its signal, and the abort + // must reach the upstream body for the pipe to end. + const ctrl = createStreamController({ provider: "ollama", model: "test" }); + + // Emits one chunk then goes silent; errors on abort like a real fetch body. + const upstream = new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode("data: hi\n\n")); + ctrl.signal.addEventListener("abort", () => controller.error(new Error("aborted")), { once: true }); + }, + }); + + let seen = null; + const out = pipeWithDisconnect( + { body: upstream }, + new TransformStream(), + ctrl, + (message) => { seen = message; return buildStreamErrorBytes(504, message, FORMATS.OPENAI); }, + 50 + ); + + const text = await readAll(out); + expect(seen).toBe("stream stall timeout"); + expect(text).toContain('"stream stall timeout"'); + expect(text).toContain("data: [DONE]"); + }); +}); diff --git a/tests/unit/responses-parallel-tool-calls.test.js b/tests/unit/responses-parallel-tool-calls.test.js new file mode 100644 index 00000000..e33b00c1 --- /dev/null +++ b/tests/unit/responses-parallel-tool-calls.test.js @@ -0,0 +1,165 @@ +// Parallel function_calls from a Responses upstream must stay on separate +// chat tool_calls indices. Regression: response/openai-responses.js attributed +// every arguments delta to the positional toolCallIndex (advanced only on +// output_item.done), so all-added-then-deltas ordering concatenated N JSON +// payloads into index 0 and clients failed with InputValidationError. +import { describe, expect, it } from "vitest"; +import "../translator/registerAll.js"; +import { openaiResponsesToOpenAIResponse } from "../../open-sse/translator/response/openai-responses.js"; +import { clampResponsesCallId, coerceResponsesOutput, MAX_RESPONSES_CALL_ID_LEN } from "../../open-sse/translator/formats/responsesApi.js"; +import { initState, translateResponse } from "../../open-sse/translator/index.js"; +import { FORMATS } from "../../open-sse/translator/formats.js"; + +const added = (id, call_id, name, type = "function_call") => ({ + type: "response.output_item.added", + item: { id, type, call_id, name, arguments: "" }, +}); +const delta = (item_id, text) => ({ + type: "response.function_call_arguments.delta", + item_id, + delta: text, +}); +const done = (id, call_id, name) => ({ + type: "response.output_item.done", + item: { id, type: "function_call", call_id, name }, +}); + +// Reassemble translated chunks the way an OpenAI client accumulator does. +function accumulate(calls, chunks) { + for (const chunk of chunks) { + if (!chunk) continue; + for (const tc of chunk.choices?.[0]?.delta?.tool_calls || []) { + const slot = (calls[tc.index] ??= { id: null, name: "", args: "" }); + if (tc.id) slot.id = tc.id; + if (tc.function?.name) slot.name = tc.function.name; + if (tc.function?.arguments) slot.args += tc.function.arguments; + } + } + return calls; +} + +function runStream(events) { + const state = {}; + const chunks = []; + for (const ev of events) { + const out = openaiResponsesToOpenAIResponse(ev, state); + if (out) chunks.push(out); + } + const flush = openaiResponsesToOpenAIResponse(null, state); + if (flush) chunks.push(flush); + return { state, chunks }; +} + +const PAYLOADS = [ + '{"file_path":"/docs/PRODUCT.md"}', + '{"file_path":"/docs/ROADMAP.md"}', + '{"file_path":"/docs/openapi.custom.yaml"}', + '{"file_path":"/docs/.gitignore"}', +]; + +function hostileOrdering() { + const events = PAYLOADS.map((_, i) => added(`fc_${i}`, `call_${i}`, "read_file")); + // Interleaved deltas AFTER all addeds — the ordering that used to merge all + // four payloads into index 0. + PAYLOADS.forEach((p, i) => events.push(delta(`fc_${i}`, p.slice(0, 20)), delta(`fc_${i}`, p.slice(20)))); + PAYLOADS.forEach((_, i) => events.push(done(`fc_${i}`, `call_${i}`, "read_file"))); + return events; +} + +describe("responses parallel tool calls keep their own index", () => { + it("all-added-then-deltas ordering yields 4 separately parseable calls", () => { + const { chunks } = runStream(hostileOrdering()); + const calls = accumulate({}, chunks); + expect(Object.keys(calls)).toHaveLength(4); + PAYLOADS.forEach((p, i) => { + expect(calls[i].id).toBe(`call_${i}`); + expect(calls[i].name).toBe("read_file"); + expect(JSON.parse(calls[i].args)).toEqual(JSON.parse(p)); + }); + }); + + it("sequential ordering still yields indices 0,1 in order", () => { + const events = [ + added("fc_0", "call_0", "read_file"), + delta("fc_0", PAYLOADS[0]), + done("fc_0", "call_0", "read_file"), + added("fc_1", "call_1", "read_file"), + delta("fc_1", PAYLOADS[1]), + done("fc_1", "call_1", "read_file"), + ]; + const { chunks } = runStream(events); + const calls = accumulate({}, chunks); + expect(Object.keys(calls)).toEqual(["0", "1"]); + expect(JSON.parse(calls[0].args)).toEqual(JSON.parse(PAYLOADS[0])); + expect(JSON.parse(calls[1].args)).toEqual(JSON.parse(PAYLOADS[1])); + }); + + it("done carrying full arguments (no deltas) emits them once", () => { + const state = {}; + const out1 = openaiResponsesToOpenAIResponse(added("fc_9", "call_9", "read_file"), state); + const out2 = openaiResponsesToOpenAIResponse({ + type: "response.output_item.done", + item: { id: "fc_9", type: "function_call", call_id: "call_9", name: "read_file", arguments: PAYLOADS[0] }, + }, state); + const calls = accumulate({}, [out1, out2]); + expect(JSON.parse(calls[0].args)).toEqual(JSON.parse(PAYLOADS[0])); + }); + + it("deltas without item_id fall back to the most recent call (legacy behavior)", () => { + const events = [ + added("fc_0", "call_0", "read_file"), + { type: "response.function_call_arguments.delta", delta: PAYLOADS[0] }, + done("fc_0", "call_0", "read_file"), + ]; + const { chunks } = runStream(events); + const calls = accumulate({}, chunks); + expect(JSON.parse(calls[0].args)).toEqual(JSON.parse(PAYLOADS[0])); + }); +}); + +describe("responses → claude end-to-end keeps parallel tool_use blocks separate", () => { + it("four read_file calls arrive as four parseable tool_use blocks", () => { + const state = initState(FORMATS.CLAUDE); + const out = []; + for (const ev of hostileOrdering()) { + for (const r of translateResponse(FORMATS.OPENAI_RESPONSES, FORMATS.CLAUDE, ev, state)) out.push(r); + } + for (const r of translateResponse(FORMATS.OPENAI_RESPONSES, FORMATS.CLAUDE, null, state)) out.push(r); + + const starts = out.filter((r) => r?.type === "content_block_start" && r?.content_block?.type === "tool_use"); + expect(starts).toHaveLength(4); + const partials = out.filter((r) => r?.delta?.type === "input_json_delta"); + expect(partials).toHaveLength(4); + const bodies = partials.map((r) => JSON.parse(r.delta.partial_json).file_path).sort(); + expect(bodies).toEqual([ + "/docs/.gitignore", + "/docs/PRODUCT.md", + "/docs/ROADMAP.md", + "/docs/openapi.custom.yaml", + ]); + }); +}); + +describe("fallback call_ids stay unique within a batch", () => { + it("same-millisecond fallbacks never collide", () => { + const ids = new Set(Array.from({ length: 50 }, () => clampResponsesCallId(undefined))); + expect(ids.size).toBe(50); + for (const id of ids) { + expect(id.startsWith("call_")).toBe(true); + expect(id.length).toBeLessThanOrEqual(MAX_RESPONSES_CALL_ID_LEN); + } + expect(new Set([clampResponsesCallId(""), clampResponsesCallId(null)]).size).toBe(2); + }); +}); + +describe("output coercion stays fail-soft on unstringifiable values", () => { + it("never throws on BigInt/circular array elements", () => { + const circular = {}; + circular.self = circular; + const input = [1n, circular, { text: "ok" }]; + expect(() => coerceResponsesOutput(input)).not.toThrow(); + const out = coerceResponsesOutput(input); + expect(typeof out).toBe("string"); + expect(out).toContain("ok"); + }); +}); diff --git a/tests/unit/system-inject.test.js b/tests/unit/system-inject.test.js index bff7c94c..fd074f2d 100644 --- a/tests/unit/system-inject.test.js +++ b/tests/unit/system-inject.test.js @@ -261,138 +261,121 @@ describe("system-inject gemini", () => { }); describe("system-inject kiro", () => { - it("updates systemPrompt and mirrored prefix of first history user preserving tail", () => { - const oldPrompt = "OLD_SYS"; + // The kiro.dev gateway rejects any body carrying a top-level `systemPrompt` + // with 400 REQUEST_BODY_INVALID, so the prompt goes into the user turn only. + it("appends to first history user, leaves systemPrompt untouched", () => { const timeCtx = "[Context: Current time is 2026-01-01T00:00:00.000Z]"; const tail = "user tail content"; - const historyUserContent = `${oldPrompt}${SEP}${timeCtx}${SEP}${tail}`; + const historyUserContent = `${timeCtx}${SEP}${tail}`; const body = { - systemPrompt: oldPrompt, conversationState: { history: [{ userInputMessage: { content: historyUserContent, modelId: "m" } }, { assistantResponseMessage: { content: "..." } }], currentMessage: { userInputMessage: { content: "current " + tail, modelId: "m" } }, }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - const next = `${oldPrompt}${SEP}${P1}`; - expect(body.systemPrompt).toBe(next); - expect(body.conversationState.history[0].userInputMessage.content).toBe(`${next}${SEP}${timeCtx}${SEP}${tail}`); + expect(body.systemPrompt).toBeUndefined(); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`${historyUserContent}${SEP}${P1}`); // currentMessage must stay untouched expect(body.conversationState.currentMessage.userInputMessage.content).toBe("current " + tail); }); - it("when no history user, updates currentMessage instead", () => { - const oldPrompt = "OLD"; - const body = { - systemPrompt: oldPrompt, - conversationState: { - history: [], - currentMessage: { userInputMessage: { content: `${oldPrompt}${SEP}tail`, modelId: "m" } }, - }, - }; - injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(`${oldPrompt}${SEP}${P1}`); - expect(body.conversationState.currentMessage.userInputMessage.content).toBe(`${oldPrompt}${SEP}${P1}${SEP}tail`); - }); - - it("empty old prompt prepends to chosen user content", () => { - const body = { - systemPrompt: "", - conversationState: { - history: [{ userInputMessage: { content: "tail hello", modelId: "m" } }], - currentMessage: { userInputMessage: { content: "cur", modelId: "m" } }, - }, - }; - injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(P1); - expect(body.conversationState.history[0].userInputMessage.content).toBe(`${P1}${SEP}tail hello`); - }); - - it("if old prompt not mirrored at head, do not alter user content", () => { + it("never writes a top-level systemPrompt, even if one is already present", () => { const body = { systemPrompt: "OLD", conversationState: { - history: [{ userInputMessage: { content: "different head content", modelId: "m" } }], + history: [{ userInputMessage: { content: "tail", modelId: "m" } }], + }, + }; + injectSystemPrompt(body, FORMATS.KIRO, P1); + expect(body.systemPrompt).toBe("OLD"); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`tail${SEP}${P1}`); + }); + + it("when no history user, updates currentMessage instead", () => { + const body = { + conversationState: { + history: [], + currentMessage: { userInputMessage: { content: "tail", modelId: "m" } }, + }, + }; + injectSystemPrompt(body, FORMATS.KIRO, P1); + expect(body.systemPrompt).toBeUndefined(); + expect(body.conversationState.currentMessage.userInputMessage.content).toBe(`tail${SEP}${P1}`); + }); + + it("empty user content becomes the prompt itself", () => { + const body = { + conversationState: { + history: [{ userInputMessage: { content: "", modelId: "m" } }], currentMessage: { userInputMessage: { content: "cur", modelId: "m" } }, }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(`OLD${SEP}${P1}`); - expect(body.conversationState.history[0].userInputMessage.content).toBe("different head content"); + expect(body.conversationState.history[0].userInputMessage.content).toBe(P1); + expect(body.conversationState.currentMessage.userInputMessage.content).toBe("cur"); }); it("exact retry idempotency for kiro", () => { - const oldPrompt = "OLD"; const body = { - systemPrompt: oldPrompt, conversationState: { - history: [{ userInputMessage: { content: `${oldPrompt}${SEP}tail`, modelId: "m" } }], + history: [{ userInputMessage: { content: "tail", modelId: "m" } }], currentMessage: { userInputMessage: { content: "cur", modelId: "m" } }, }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - const after1 = JSON.parse(JSON.stringify(body)); + const after1 = body.conversationState.history[0].userInputMessage.content; injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(after1.systemPrompt); - expect(body.conversationState.history[0].userInputMessage.content).toBe(after1.conversationState.history[0].userInputMessage.content); - // different prompt both apply + expect(body.conversationState.history[0].userInputMessage.content).toBe(after1); + // different prompt both apply, in injection order injectSystemPrompt(body, FORMATS.KIRO, P2); - expect(body.systemPrompt).toBe(`${oldPrompt}${SEP}${P1}${SEP}${P2}`); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`tail${SEP}${P1}${SEP}${P2}`); }); it("preserves non-enumerable _kiroUpstreamModel", () => { const body = { - systemPrompt: "OLD", - conversationState: { history: [{ userInputMessage: { content: "OLD" + SEP + "tail", modelId: "m" } }], currentMessage: { userInputMessage: { content: "OLD" + SEP + "tail2", modelId: "m" } } }, + conversationState: { history: [{ userInputMessage: { content: "tail", modelId: "m" } }], currentMessage: { userInputMessage: { content: "tail2", modelId: "m" } } }, }; Object.defineProperty(body, "_kiroUpstreamModel", { value: "m", enumerable: false }); injectSystemPrompt(body, FORMATS.KIRO, P1); expect(body._kiroUpstreamModel).toBe("m"); expect(Object.getOwnPropertyDescriptor(body, "_kiroUpstreamModel").enumerable).toBe(false); }); + + it("frozen user message fails open without throwing or half-writing", () => { + const body = { + conversationState: { + history: [{ userInputMessage: Object.freeze({ content: "tail", modelId: "m" }) }], + }, + }; + expect(() => injectSystemPrompt(body, FORMATS.KIRO, P1)).not.toThrow(); + expect(body.systemPrompt).toBeUndefined(); + expect(body.conversationState.history[0].userInputMessage.content).toBe("tail"); + }); }); describe("system-inject regression fixes", () => { it("kiro partial mutation converges on retry after transient content write failure", () => { - const oldPrompt = "OLD"; let failNextWrite = true; - const um = { content: `${oldPrompt}${SEP}tail`, modelId: "m" }; + const um = { content: "tail", modelId: "m" }; const proxiedUm = new Proxy(um, { set(t, p, v) { if (p === "content" && failNextWrite) { failNextWrite = false; throw new Error("transient"); } t[p] = v; return true; }, }); - const body = { - systemPrompt: oldPrompt, - conversationState: { - history: [{ userInputMessage: proxiedUm }], - }, - }; + const body = { conversationState: { history: [{ userInputMessage: proxiedUm }] } }; injectSystemPrompt(body, FORMATS.KIRO, P1); - // first pass rolled back atomically — nothing half-applied - expect(body.systemPrompt).toBe(oldPrompt); - expect(um.content).toBe(`${oldPrompt}${SEP}tail`); + // nothing half-applied + expect(um.content).toBe("tail"); + expect(body.systemPrompt).toBeUndefined(); // retry converges injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(`${oldPrompt}${SEP}${P1}`); - expect(um.content).toBe(`${oldPrompt}${SEP}${P1}${SEP}tail`); - }); - - it("kiro rolls back systemPrompt when user content write fails (atomicity)", () => { - const oldPrompt = "OLD"; - const body = { - systemPrompt: oldPrompt, - conversationState: { - history: [{ userInputMessage: Object.freeze({ content: `${oldPrompt}${SEP}tail`, modelId: "m" }) }], - }, - }; - injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.systemPrompt).toBe(oldPrompt); + expect(um.content).toBe(`tail${SEP}${P1}`); }); it("kiro shape gate: stray conversationState without history/currentMessage does not hijack chat body", () => { - const body = { messages: [{ role: ROLE.SYSTEM, content: "hello" }], systemPrompt: "", conversationState: {} }; + const body = { messages: [{ role: ROLE.SYSTEM, content: "hello" }], conversationState: {} }; injectSystemPrompt(body, FORMATS.OPENAI, P1); expect(body.messages[0].content).toBe(`hello${SEP}${P1}`); }); @@ -409,15 +392,14 @@ describe("system-inject regression fixes", () => { expect(body.instructions).toBe(`You are RULE follower${SEP}RULE`); }); - it("kiro empty-old prepend fires when prompt appears mid-tail only", () => { + it("substring occurrence does not suppress kiro injection", () => { const body = { - systemPrompt: "", conversationState: { history: [{ userInputMessage: { content: `some ${P1} here`, modelId: "m" } }], }, }; injectSystemPrompt(body, FORMATS.KIRO, P1); - expect(body.conversationState.history[0].userInputMessage.content).toBe(`${P1}${SEP}some ${P1} here`); + expect(body.conversationState.history[0].userInputMessage.content).toBe(`some ${P1} here${SEP}${P1}`); }); }); diff --git a/tests/unit/usage-dispatch.test.js b/tests/unit/usage-dispatch.test.js index 5b86ee9e..ead63892 100644 --- a/tests/unit/usage-dispatch.test.js +++ b/tests/unit/usage-dispatch.test.js @@ -16,7 +16,7 @@ const SUPPORTED = [ "github", "gemini-cli", "antigravity", "claude", "codex", "kiro", "qoder", "iflow", "ollama", "glm", "glm-cn", "minimax", "minimax-cn", "vercel-ai-gateway", "grok-cli", "kimi", - "deepseek", "zed", + "deepseek", "opencode-go", "zed", "commandcode", ]; describe("usage dispatch", () => { diff --git a/tests/unit/video-providers.test.js b/tests/unit/video-providers.test.js new file mode 100644 index 00000000..bfa26589 --- /dev/null +++ b/tests/unit/video-providers.test.js @@ -0,0 +1,296 @@ +/** + * Unit tests for the OpenRouter + Vertex (Veo) video adapters. + * + * Covers: + * - registry wiring (videoConfig, video serviceKind, video-kind models) + * - OpenRouter: POST to the collection root, GET poll, verbatim passthrough + * - Vertex: predictLongRunning body translation, fetchPredictOperation polling, + * operation-name round-trip through the job id, response mapping + * - xAI default path is unchanged by the adapter hook + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; + +vi.mock("open-sse/services/tokenRefresh.js", async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, refreshTokenByProvider: vi.fn(), refreshVertexToken: vi.fn() }; +}); + +import { handleVideoProxyCore, getVideoConfig } from "open-sse/handlers/videoCore.js"; +import { refreshVertexToken } from "open-sse/services/tokenRefresh.js"; +import { PROVIDER_MEDIA, PROVIDER_MODELS } from "open-sse/providers/index.js"; + +const originalFetch = global.fetch; +const jsonResponse = (body, status = 200) => + new Response(JSON.stringify(body), { status, headers: { "Content-Type": "application/json" } }); + +// Vertex operation names are resource paths; the adapter base64url-encodes them. +const OPERATION_NAME = + "projects/proj-1/locations/us-central1/publishers/google/models/veo-3.1-generate-preview/operations/op-abc"; +const JOB_ID = Buffer.from(OPERATION_NAME, "utf8").toString("base64url"); + +describe("registry wiring", () => { + it("exposes videoConfig + video serviceKind for openrouter and vertex", () => { + expect(getVideoConfig("openrouter").baseUrl).toBe("https://openrouter.ai/api/v1/videos"); + expect(getVideoConfig("vertex").baseUrl).toBe("https://aiplatform.googleapis.com"); + expect(PROVIDER_MEDIA.openrouter.serviceKinds).toContain("video"); + expect(PROVIDER_MEDIA.vertex.serviceKinds).toContain("video"); + }); + + it("registers video-kind models on both providers", () => { + const or = PROVIDER_MODELS.openrouter.find((m) => m.id === "google/veo-3.1"); + const vx = PROVIDER_MODELS.vertex.find((m) => m.id === "veo-3.1-generate-preview"); + expect(or?.kind).toBe("video"); + expect(vx?.kind).toBe("video"); + }); +}); + +describe("openrouter video adapter", () => { + beforeEach(() => { global.fetch = vi.fn(); }); + afterEach(() => { global.fetch = originalFetch; }); + + it("POSTs creation to the collection root (no /generations suffix)", async () => { + global.fetch.mockResolvedValueOnce(jsonResponse({ id: "job-1", status: "pending" })); + + const raw = '{"model":"google/veo-3.1","prompt":"a paper boat"}'; + const result = await handleVideoProxyCore({ + provider: "openrouter", + action: "generations", + rawBody: raw, + contentType: "application/json", + credentials: { apiKey: "sk-or-key" }, + }); + + expect(result.success).toBe(true); + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe("https://openrouter.ai/api/v1/videos"); + expect(init.method).toBe("POST"); + expect(init.body).toBe(raw); // verbatim + expect(init.headers.Authorization).toBe("Bearer sk-or-key"); + expect(init.headers["HTTP-Referer"]).toBe("https://endpoint-proxy.local"); + expect(await result.response.json()).toEqual({ id: "job-1", status: "pending" }); + }); + + it("polls GET /videos/{id} and passes the payload through verbatim", async () => { + const payload = { id: "job-1", status: "completed", unsigned_urls: ["https://cdn/v.mp4"] }; + global.fetch.mockResolvedValueOnce(jsonResponse(payload)); + + const result = await handleVideoProxyCore({ + provider: "openrouter", + requestId: "job-1", + credentials: { apiKey: "sk-or-key" }, + }); + + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe("https://openrouter.ai/api/v1/videos/job-1"); + expect(init.method).toBe("GET"); + expect(await result.response.json()).toEqual(payload); + }); + + it("rejects unsupported actions before any upstream call (no billable job)", async () => { + const result = await handleVideoProxyCore({ + provider: "openrouter", + action: "extensions", + rawBody: "{}", + contentType: "application/json", + credentials: { apiKey: "sk-or-key" }, + }); + + expect(result.success).toBe(false); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + }); +}); + +describe("vertex (veo) video adapter", () => { + beforeEach(() => { + global.fetch = vi.fn(); + refreshVertexToken.mockReset(); + }); + afterEach(() => { global.fetch = originalFetch; }); + + const saJson = JSON.stringify({ + type: "service_account", + client_email: "sa@proj-1.iam.gserviceaccount.com", + private_key: "-----BEGIN PRIVATE KEY-----\nx\n-----END PRIVATE KEY-----\n", + project_id: "proj-1", + }); + + it("translates the create body to predictLongRunning and returns a poll-able job id", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce(jsonResponse({ name: OPERATION_NAME })); + + const result = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ + model: "veo-3.1-generate-preview", + prompt: "a neon city", + duration: 8, + aspect_ratio: "16:9", + resolution: "720p", + n: 1, + }), + contentType: "application/json", + credentials: { apiKey: saJson }, + }); + + expect(result.success).toBe(true); + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe( + "https://aiplatform.googleapis.com/v1/projects/proj-1/locations/us-central1/publishers/google/models/veo-3.1-generate-preview:predictLongRunning" + ); + expect(init.headers.Authorization).toBe("Bearer vertex-tok"); + expect(JSON.parse(init.body)).toEqual({ + instances: [{ prompt: "a neon city" }], + parameters: { sampleCount: 1, durationSeconds: 8, aspectRatio: "16:9", resolution: "720p" }, + }); + + // Response is mapped onto the async-job shape clients already poll. + expect(await result.response.json()).toEqual({ + id: JOB_ID, + request_id: JOB_ID, + status: "pending", + }); + }); + + it("maps a data-URL image onto the Vertex image instance", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce(jsonResponse({ name: OPERATION_NAME })); + + await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ + model: "veo-3.1-generate-preview", + prompt: "animate this", + image: "data:image/png;base64,AAAB", + }), + contentType: "application/json", + credentials: { apiKey: saJson }, + }); + + expect(JSON.parse(global.fetch.mock.calls[0][1].body).instances[0].image).toEqual({ + bytesBase64Encoded: "AAAB", + mimeType: "image/png", + }); + }); + + it("polls via fetchPredictOperation and maps a completed operation", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce( + jsonResponse({ + name: OPERATION_NAME, + done: true, + response: { videos: [{ gcsUri: "gs://bucket/v.mp4", mimeType: "video/mp4" }] }, + }) + ); + + const result = await handleVideoProxyCore({ + provider: "vertex", + requestId: JOB_ID, + credentials: { apiKey: saJson }, + }); + + const [url, init] = global.fetch.mock.calls[0]; + expect(url).toBe( + "https://aiplatform.googleapis.com/v1/projects/proj-1/locations/us-central1/publishers/google/models/veo-3.1-generate-preview:fetchPredictOperation" + ); + expect(init.method).toBe("POST"); // Vertex polls with POST, not GET + expect(JSON.parse(init.body)).toEqual({ operationName: OPERATION_NAME }); + + expect(await result.response.json()).toEqual({ + id: JOB_ID, + request_id: JOB_ID, + status: "completed", + video: { url: "gs://bucket/v.mp4", b64_json: null, mime_type: "video/mp4" }, + videos: [{ url: "gs://bucket/v.mp4", b64_json: null, mime_type: "video/mp4" }], + }); + }); + + it("maps a failed operation to status failed", async () => { + refreshVertexToken.mockResolvedValueOnce({ accessToken: "vertex-tok" }); + global.fetch.mockResolvedValueOnce( + jsonResponse({ name: OPERATION_NAME, done: true, error: { code: 3, message: "bad prompt" } }) + ); + + const result = await handleVideoProxyCore({ + provider: "vertex", + requestId: JOB_ID, + credentials: { apiKey: saJson }, + }); + + const body = await result.response.json(); + expect(body.status).toBe("failed"); + expect(body.error.message).toBe("bad prompt"); + }); + + it("rejects missing project id and raw API keys before any upstream call", async () => { + const noProject = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ model: "veo-3.1-generate-preview", prompt: "x" }), + contentType: "application/json", + credentials: { apiKey: "AIzaRawKey" }, + }); + expect(noProject.success).toBe(false); + expect(noProject.status).toBe(400); + + const noToken = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ model: "veo-3.1-generate-preview", prompt: "x" }), + contentType: "application/json", + credentials: { apiKey: "AIzaRawKey", providerSpecificData: { projectId: "proj-1" } }, + }); + expect(noToken.success).toBe(false); + expect(noToken.status).toBe(400); + + expect(global.fetch).not.toHaveBeenCalled(); + }); + + it("rejects an invalid job id without calling upstream", async () => { + refreshVertexToken.mockResolvedValue({ accessToken: "vertex-tok" }); + const result = await handleVideoProxyCore({ + provider: "vertex", + requestId: Buffer.from("not-an-operation", "utf8").toString("base64url"), + credentials: { apiKey: saJson }, + }); + expect(result.success).toBe(false); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + }); + + // A base64url id decodes to arbitrary bytes, so a crafted one used to splice a + // path traversal into the fetch URL while the Authorization header stayed on. + it("rejects job ids that decode outside the projects/…/operations/ shape", async () => { + refreshVertexToken.mockResolvedValue({ accessToken: "vertex-tok" }); + const jid = (s) => Buffer.from(s, "utf8").toString("base64url"); + + for (const id of [ + jid("../../evil"), + jid("projects/p/locations/l/publishers/google/models/m/operations/../../x"), + jid("../../evil/operations/op"), + "!!!not-base64!!!", + `${JOB_ID}=`, + `${JOB_ID}\n`, + ]) { + const result = await handleVideoProxyCore({ provider: "vertex", requestId: id, credentials: { apiKey: saJson } }); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + } + }); + + it("rejects a model id carrying path separators", async () => { + refreshVertexToken.mockResolvedValue({ accessToken: "vertex-tok" }); + const result = await handleVideoProxyCore({ + provider: "vertex", + action: "generations", + rawBody: JSON.stringify({ model: "../../evil", prompt: "x" }), + contentType: "application/json", + credentials: { apiKey: saJson }, + }); + expect(result.status).toBe(400); + expect(global.fetch).not.toHaveBeenCalled(); + }); +}); diff --git a/tests/unit/xai-video-handler.test.js b/tests/unit/xai-video-handler.test.js index 563ee08c..a18411b4 100644 --- a/tests/unit/xai-video-handler.test.js +++ b/tests/unit/xai-video-handler.test.js @@ -29,6 +29,7 @@ vi.mock("@/sse/services/auth.js", () => authMocks); vi.mock("@/sse/services/tokenRefresh.js", () => tokenMocks); vi.mock("@/lib/localDb", () => ({ getSettings: vi.fn(async () => ({ requireApiKey: false })), + getProviderConnectionById: vi.fn(async () => ({ id: "conn-5", provider: "xai" })), getComboByName: vi.fn(async () => null), getModelAliases: vi.fn(async () => ({})), getProviderNodes: vi.fn(async () => []), diff --git a/tests/unit/xiaomi-mimo-executor.test.js b/tests/unit/xiaomi-mimo-executor.test.js new file mode 100644 index 00000000..10c6a520 --- /dev/null +++ b/tests/unit/xiaomi-mimo-executor.test.js @@ -0,0 +1,80 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; +import { XiaomiMimoExecutor, __test__ } from "../../open-sse/executors/xiaomi-mimo.js"; +import { getExecutor } from "../../open-sse/executors/index.js"; + +const { bareModel, COOKIE_KEY } = __test__; + +const OPENAI_T = { runtimeTransport: { format: "openai", baseUrl: "https://api.xiaomimimo.com/v1/chat/completions" } }; +const CLAUDE_T = { runtimeTransport: { format: "claude", baseUrl: "https://api.xiaomimimo.com/anthropic/v1/messages" } }; + +describe("xiaomi-mimo executor", () => { + let ex; + beforeEach(() => { + ex = new XiaomiMimoExecutor(); + }); + + it("is registered for xiaomi-mimo", () => { + expect(getExecutor("xiaomi-mimo")).toBeInstanceOf(XiaomiMimoExecutor); + }); + + it("routes Preview models to the account-service route regardless of transport", () => { + const expected = "https://mimo-server-cn.xiaomimimo.com/api/route/chat/completions"; + expect(ex.buildUrl("mimo-x-pro-preview", true, 0, OPENAI_T)).toBe(expected); + expect(ex.buildUrl("mimo-x-pro-preview", true, 0, CLAUDE_T)).toBe(expected); + // body.model arrives as `xiaomi/` via upstreamModelId + expect(ex.buildUrl("xiaomi/mimo-x-flash-preview", true, 0, OPENAI_T)).toBe(expected); + }); + + it("keeps the sourceFormat-matched endpoint for cloud models", () => { + // Regression: a Claude client must reach /anthropic/v1/messages, not /v1/chat/completions. + expect(ex.buildUrl("mimo-v2.5-pro", true, 0, CLAUDE_T)).toBe(CLAUDE_T.runtimeTransport.baseUrl); + expect(ex.buildUrl("mimo-v2.5-pro", true, 0, OPENAI_T)).toBe(OPENAI_T.runtimeTransport.baseUrl); + }); + + it("authenticates Preview calls with the account cookie", () => { + const headers = ex.buildHeaders({ [COOKIE_KEY]: "serviceToken=abc", accessToken: "sk-x" }, true, "u", "mimo-x-pro-preview"); + expect(headers.Cookie).toBe("serviceToken=abc"); + expect(headers.Authorization).toBeUndefined(); + }); + + it("authenticates cloud calls with the bearer key", () => { + const headers = ex.buildHeaders({ accessToken: "sk-x" }, true, "u", "mimo-v2.5-pro"); + expect(headers.Authorization).toBe("Bearer sk-x"); + expect(headers.Cookie).toBeUndefined(); + }); + + it("fails fast when a Preview call has no account session", async () => { + await expect( + ex.execute({ model: "mimo-x-pro-preview", body: {}, stream: true, credentials: {}, log: null }), + ).rejects.toThrow(/account session unavailable/); + }); + + it("flattens content-part arrays to plain strings", () => { + const out = ex.transformRequest( + "mimo-x-pro-preview", + { messages: [{ role: "user", content: [{ type: "text", text: "a" }, { type: "text", text: "b" }] }] }, + true, + {}, + ); + expect(out.messages[0].content).toBe("ab"); + }); + + it("applies Preview defaults without overriding explicit values", () => { + const body = { messages: [{ role: "user", content: "hi" }], temperature: 0.2 }; + const out = ex.transformRequest("mimo-x-pro-preview", body, true, {}); + expect(out.temperature).toBe(0.2); // caller's value kept + expect(out.top_p).toBe(0.95); // default filled in + expect(out.max_tokens).toBe(4096); + }); + + it("leaves cloud bodies free of Preview defaults", () => { + const out = ex.transformRequest("mimo-v2.5-pro", { messages: [{ role: "user", content: "hi" }] }, true, {}); + expect(out.thinking).toBeUndefined(); + expect(out.max_tokens).toBeUndefined(); + }); + + it("strips a provider/model prefix when testing preview ids", () => { + expect(bareModel("xiaomi/mimo-x-pro-preview")).toBe("mimo-x-pro-preview"); + expect(bareModel("mimo-x-pro-preview")).toBe("mimo-x-pro-preview"); + }); +}); diff --git a/tests/unit/xiaomi-mimo-oauth-proxy.test.js b/tests/unit/xiaomi-mimo-oauth-proxy.test.js new file mode 100644 index 00000000..c74573ff --- /dev/null +++ b/tests/unit/xiaomi-mimo-oauth-proxy.test.js @@ -0,0 +1,54 @@ +/** + * Regression: the xiaomi-mimo OAuth session store must not retain sessions + * once the callback listener is down. + * + * Each /authorize registers a session holding an X25519 private key, keyed by a + * fresh state. Unlike trae/windsurf/zed (singleton session) this is a Map, so + * without an explicit clear every login attempt would leak a private key for + * the whole process lifetime. + */ +import { describe, it, expect } from "vitest"; +import { + registerXiaomiMimoSession, + getXiaomiMimoSessionStatus, + clearXiaomiMimoSession, + stopXiaomiMimoProxy, +} from "../../src/lib/oauth/utils/server.js"; + +const KEY = Buffer.from("x25519-private-key-material"); + +describe("xiaomi-mimo OAuth session store", () => { + it("drops pending sessions when the proxy stops", () => { + registerXiaomiMimoSession({ state: "s1", privateKeyDer: KEY }); + expect(getXiaomiMimoSessionStatus("s1")).not.toBeNull(); + + stopXiaomiMimoProxy(); + + expect(getXiaomiMimoSessionStatus("s1")).toBeNull(); + }); + + it("drops every session, not just the last one", () => { + registerXiaomiMimoSession({ state: "a", privateKeyDer: KEY }); + registerXiaomiMimoSession({ state: "b", privateKeyDer: KEY }); + registerXiaomiMimoSession({ state: "c", privateKeyDer: KEY }); + + stopXiaomiMimoProxy(); + + for (const s of ["a", "b", "c"]) { + expect(getXiaomiMimoSessionStatus(s)).toBeNull(); + } + }); + + it("ignores registrations with a missing state or key", () => { + expect(registerXiaomiMimoSession({ state: "", privateKeyDer: KEY })).toBe(false); + expect(registerXiaomiMimoSession({ state: "s", privateKeyDer: null })).toBe(false); + }); + + it("never exposes the private key to callers", () => { + registerXiaomiMimoSession({ state: "s1", privateKeyDer: KEY }); + const view = getXiaomiMimoSessionStatus("s1"); + expect(view).toEqual({ status: "pending", result: null, error: null }); + expect(JSON.stringify(view)).not.toContain("privateKeyDer"); + clearXiaomiMimoSession("s1"); + }); +}); diff --git a/tests/unit/xiaomi-mimo-oauth-session.test.js b/tests/unit/xiaomi-mimo-oauth-session.test.js new file mode 100644 index 00000000..3aee1cc2 --- /dev/null +++ b/tests/unit/xiaomi-mimo-oauth-session.test.js @@ -0,0 +1,152 @@ +/** + * Regression: the poll-status/exchange session lifecycle for xiaomi-mimo. + * + * The original PR cleared the session inside poll-status, so the client's + * following POST /exchange always saw a missing session and returned 400 — + * the whole browser-OAuth fallback was dead. These tests pin the contract: + * - a finished session survives /poll-status until /exchange consumes it + * - a failed session is cleaned up by /poll-status itself + */ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +vi.mock("next/server", () => ({ + NextResponse: { + json: (body, init) => ({ + status: init?.status || 200, + body, + json: async () => body, + }), + }, +})); + +vi.mock("@/lib/oauth/providers", () => ({ + getProvider: vi.fn(), + generateAuthData: vi.fn(), + exchangeTokens: vi.fn(), + requestDeviceCode: vi.fn(), + pollForToken: vi.fn(), +})); + +vi.mock("@/models", () => ({ + createProviderConnection: vi.fn(async (d) => ({ id: "conn-1", ...d })), +})); + +vi.mock("open-sse/shared/mimoAccount.js", () => ({ + readDesktopPassToken: vi.fn(async () => ({ passToken: "pt-abc", userId: "u1", cUserId: "c1" })), +})); + +vi.mock("@/lib/oauth/utils/ideDetect", () => ({ detectIdeInstalled: vi.fn() })); + +// Session store backing the mocked OAuth server helpers, so the test can assert +// on real lifecycle transitions rather than on call counts alone. +const sessions = new Map(); +const stopped = { count: 0 }; + +vi.mock("@/lib/oauth/utils/server", () => { + const notUsed = () => { throw new Error("unexpected helper"); }; + const noop = () => {}; + return { + startCodexProxy: notUsed, stopCodexProxy: noop, registerCodexSession: noop, + getCodexSessionStatus: () => null, clearCodexSession: noop, + startXaiProxy: notUsed, stopXaiProxy: noop, registerXaiSession: noop, + getXaiSessionStatus: () => null, clearXaiSession: noop, + startTraeProxy: notUsed, stopTraeProxy: noop, registerTraeSession: noop, + getTraeSessionStatus: () => null, clearTraeSession: noop, + startWindsurfProxy: notUsed, stopWindsurfProxy: noop, registerWindsurfSession: noop, + getWindsurfSessionStatus: () => null, clearWindsurfSession: noop, + startZedProxy: notUsed, stopZedProxy: noop, registerZedSession: noop, + getZedSessionStatus: () => null, clearZedSession: noop, + startXiaomiMimoProxy: notUsed, + stopXiaomiMimoProxy: () => { stopped.count += 1; }, + registerXiaomiMimoSession: () => {}, + getXiaomiMimoSessionStatus: (state) => { + const s = sessions.get(state); + return s ? { status: s.status, result: s.result || null, error: s.error || null } : null; + }, + clearXiaomiMimoSession: (state) => { sessions.delete(state); }, + }; +}); + +const { GET, POST } = await import("../../src/app/api/oauth/[provider]/[action]/route.js"); + +const get = (action, state) => + GET(new Request(`http://localhost/api/oauth/xiaomi-mimo/${action}?state=${state}`), { + params: Promise.resolve({ provider: "xiaomi-mimo", action }), + }); + +const exchange = (state) => + POST( + new Request("http://localhost/api/oauth/xiaomi-mimo/exchange", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ state }), + }), + { params: Promise.resolve({ provider: "xiaomi-mimo", action: "exchange" }) }, + ); + +describe("xiaomi-mimo OAuth session lifecycle", () => { + beforeEach(() => { + sessions.clear(); + stopped.count = 0; + }); + + it("keeps a finished session alive so /exchange can consume it", async () => { + sessions.set("st1", { status: "done", result: { uid: "u1", accessToken: "sk-x", baseUrl: "https://api.xiaomimimo.com/v1" } }); + + const poll = await get("poll-status", "st1"); + expect(poll.status).toBe(200); + expect(await poll.json()).toMatchObject({ status: "done" }); + + // The bug: this used to be gone, making /exchange always 400. + expect(sessions.has("st1")).toBe(true); + + const res = await exchange("st1"); + expect(res.status).toBe(200); + expect((await res.json()).success).toBe(true); + }); + + it("clears the session once /exchange consumed it", async () => { + sessions.set("st1", { status: "done", result: { uid: "u1", accessToken: "sk-x" } }); + await exchange("st1"); + expect(sessions.has("st1")).toBe(false); + }); + + it("cleans up a failed session in poll-status and stops the proxy", async () => { + sessions.set("st2", { status: "error", error: "Could not decrypt with any pending session key" }); + + const poll = await get("poll-status", "st2"); + expect(await poll.json()).toMatchObject({ status: "error" }); + + expect(sessions.has("st2")).toBe(false); + expect(stopped.count).toBe(1); + }); + + it("persists the Desktop passToken onto the connection (Preview models need it)", async () => { + const { createProviderConnection } = await import("@/models"); + sessions.set("st3", { status: "done", result: { uid: "u1", accessToken: "sk-x" } }); + + await exchange("st3"); + + const arg = createProviderConnection.mock.calls.at(-1)[0]; + expect(arg.provider).toBe("xiaomi-mimo"); + expect(arg.providerSpecificData.mimoPassToken).toBe("pt-abc"); + expect(arg.providerSpecificData.mimoUserId).toBe("u1"); + }); + + it("still reports unknown for an unregistered state", async () => { + const poll = await get("poll-status", "nope"); + expect(await poll.json()).toEqual({ status: "unknown" }); + }); + + it("rejects /exchange without a state", async () => { + const res = await POST( + new Request("http://localhost/api/oauth/xiaomi-mimo/exchange", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({}), + }), + { params: Promise.resolve({ provider: "xiaomi-mimo", action: "exchange" }) }, + ); + expect(res.status).toBe(400); + }); +}); diff --git a/tests/unit/zed-completions-wire.test.js b/tests/unit/zed-completions-wire.test.js new file mode 100644 index 00000000..5be982ae --- /dev/null +++ b/tests/unit/zed-completions-wire.test.js @@ -0,0 +1,121 @@ +// Zed completions wire acceptance: the `provider` field of POST /completions +// must use cloud.zed.dev's exact wire values (anthropic/open_ai/google/x_ai), +// and the Zed Gemini path must not carry the shared translator's +// safetySettings (Zed's hosted Gemini backend speaks the Vertex safety +// vocabulary, not the public-Gemini enums). +import { describe, it, expect, beforeEach, vi } from "vitest"; + +vi.mock("open-sse/shared/zedAuth.js", async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + resolveZedModels: vi.fn(), + zedLlmFetch: vi.fn(), + }; +}); + +import { + resolveZedModels, + zedLlmFetch, +} from "open-sse/shared/zedAuth.js"; +import ZedExecutor from "open-sse/executors/zed.js"; + +function catalogFor(entries) { + const rawById = new Map(entries); + return { rawById, models: [] }; +} + +function mockCatalogFetch(captured) { + zedLlmFetch.mockImplementation(async (credentials, path, options) => { + captured.body = JSON.parse(options.fetchOptions.body); + return new Response("upstream-error-stub", { status: 500 }); + }); +} + +function makeExecutor() { + const executor = new ZedExecutor(); + executor.config = {}; + return executor; +} + +const CHAT_BODY = { messages: [{ role: "user", content: "hi" }] }; + +beforeEach(() => { + vi.clearAllMocks(); +}); + +describe("wire provider enum", () => { + it.each([ + ["Anthropic", "anthropic"], + ["anthropic", "anthropic"], + ["OpenAi", "open_ai"], + ["open_ai", "open_ai"], + ["Google", "google"], + ["gemini", "google"], + ["XAi", "x_ai"], + ["x_ai", "x_ai"], + ])("catalog provider %j normalizes to wire %j", async (catalogValue, wire) => { + resolveZedModels.mockResolvedValue(catalogFor([["m", { provider: catalogValue }]])); + const executor = makeExecutor(); + const { provider } = await executor.resolveModel("m", {}, null, null); + expect(provider).toBe(wire); + }); + + it("infers wire provider from the model id when the catalog is unavailable", async () => { + resolveZedModels.mockRejectedValue(new Error("catalog down")); + const executor = makeExecutor(); + const log = { warn: vi.fn() }; + expect((await executor.resolveModel("claude-opus-x", {}, null, log)).provider).toBe("anthropic"); + expect((await executor.resolveModel("gemini-3-x", {}, null, log)).provider).toBe("google"); + expect((await executor.resolveModel("grok-4-x", {}, null, log)).provider).toBe("x_ai"); + expect((await executor.resolveModel("gpt-5-x", {}, null, log)).provider).toBe("open_ai"); + }); +}); + +describe("completion payload shaping", () => { + it("sends wire provider values per model family", async () => { + resolveZedModels.mockImplementation(async () => catalogFor([ + ["claude-x", { provider: "anthropic" }], + ["gpt-x", { provider: "open_ai" }], + ["gemini-x", { provider: "google" }], + ["grok-x", { provider: "x_ai" }], + ])); + const captured = {}; + mockCatalogFetch(captured); + const executor = makeExecutor(); + + for (const [model, wire] of [ + ["claude-x", "anthropic"], + ["gpt-x", "open_ai"], + ["gemini-x", "google"], + ["grok-x", "x_ai"], + ]) { + await executor.execute({ model, body: { ...CHAT_BODY }, stream: false, credentials: {} }); + expect(captured.body.provider).toBe(wire); + expect(captured.body.model).toBe(model); + } + }); + + it("strips safetySettings on the Zed Gemini path only", async () => { + resolveZedModels.mockImplementation(async () => catalogFor([ + ["gemini-x", { provider: "google" }], + ["claude-x", { provider: "anthropic" }], + ])); + const captured = {}; + mockCatalogFetch(captured); + const executor = makeExecutor(); + + await executor.execute({ model: "gemini-x", body: { ...CHAT_BODY }, stream: false, credentials: {} }); + expect(captured.body.provider).toBe("google"); + expect(captured.body.provider_request).not.toHaveProperty("safetySettings"); + + // Sanity: the shared translator still emits safetySettings — the removal + // happens in the Zed executor, not in shared/native Gemini behavior. + const { openaiToGeminiRequest } = await import( + "open-sse/translator/request/openai-to-gemini.js" + ); + expect(openaiToGeminiRequest("gemini-x", { ...CHAT_BODY }, true)).toHaveProperty( + "safetySettings", + ); + }); +}); diff --git a/tests/unit/zed-live-models.test.js b/tests/unit/zed-live-models.test.js new file mode 100644 index 00000000..1b999044 --- /dev/null +++ b/tests/unit/zed-live-models.test.js @@ -0,0 +1,165 @@ +// Route-level acceptance for the Zed live-model wiring: +// GET /api/providers/[connectionId]/models → resolveZedModels → UI rows +// RUN WITH AN ISOLATED DB: DATA_DIR=$(mktemp -d) npx vitest run ... +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import { GET } from "@/app/api/providers/[id]/models/route.js"; +import { createProviderConnection } from "@/models/index.js"; + +// Transport stub BELOW resolveZedModels: proxyAwareFetch captures the native +// fetch at import time, so stubbing globalThis.fetch cannot intercept it. +// Mock the module instead; untouched hosts pass through to native fetch. +const stub = vi.hoisted(() => { + const nativeFetch = globalThis.fetch.bind(globalThis); + return { mode: "ok", calls: [], nativeFetch }; +}); +vi.mock("open-sse/utils/proxyFetch.js", () => ({ + proxyAwareFetch: async (url, options) => { + const u = String(url); + stub.calls.push(u); + if (u.includes("cloud.zed.dev/client/users/me")) { + return Response.json({ default_organization_id: "org-1" }); + } + if (u.includes("cloud.zed.dev/client/llm_tokens")) { + return Response.json({ token: "llm-token" }); + } + if (u.includes("cloud.zed.dev/models")) { + if (stub.mode === "error") return new Response("boom", { status: 500 }); + if (stub.mode === "empty") return Response.json({ models: [] }); + return Response.json(stub.catalog); + } + return stub.nativeFetch(url, options); + }, + default: async (url, options) => stub.nativeFetch(url, options), +})); + +stub.catalog = { + models: [ + { + id: "claude-opus-4-live", + display_name: "Claude Opus Live", + provider: "anthropic", + max_token_count: 200000, + max_output_tokens: 32000, + supports_tools: true, + supports_images: true, + supports_thinking: true, + is_disabled: false, + }, + { + id: "gpt-live", + display_name: "GPT Live", + provider: "openai", + max_token_count: 128000, + max_output_tokens: 16384, + supports_tools: true, + is_disabled: false, + }, + { + id: "retired-model", + display_name: "Retired", + provider: "openai", + is_disabled: true, + }, + ], + default_model: "claude-opus-4-live", +}; + +beforeEach(() => { + stub.mode = "ok"; + stub.calls.length = 0; +}); +afterEach(() => { + vi.restoreAllMocks(); +}); + +async function seedZed(n) { + return createProviderConnection({ + provider: "zed", + authType: "oauth", + accessToken: `tok-live-${n}-${Date.now()}`, + email: `zed-live-${n}-${Date.now()}@example.com`, + providerSpecificData: { userId: `u-${n}`, systemId: `sys-${n}` }, + testStatus: "active", + }); +} + +async function getModels(connectionId) { + const req = new Request(`http://localhost/api/providers/${connectionId}/models`); + return GET(req, { params: Promise.resolve({ id: connectionId }) }); +} + +describe("criterion 1+2 — active connection + live catalog → models with metadata", () => { + it("returns enabled models with preserved metadata, no secrets", async () => { + const conn = await seedZed("m1"); + const res = await getModels(conn.id); + expect(res.status).toBe(200); + const data = await res.json(); + expect(data.models.map((m) => m.id).sort()).toEqual(["claude-opus-4-live", "gpt-live"]); + const opus = data.models.find((m) => m.id === "claude-opus-4-live"); + expect(opus.name).toBe("Claude Opus Live"); + expect(opus.contextLength).toBe(200000); + expect(opus.maxOutputTokens).toBe(32000); + expect(opus.supportsTools).toBe(true); + expect(opus.supportsImages).toBe(true); + expect(opus.supportsThinking).toBe(true); + // Credentials must never leak into the client response. + expect(JSON.stringify(data)).not.toContain(conn.accessToken); + expect(JSON.stringify(data)).not.toContain("tok-live"); + }); +}); + +describe("criterion 4 — disabled models excluded", () => { + it("is_disabled entries never reach the UI", async () => { + const conn = await seedZed("m2"); + const data = await (await getModels(conn.id)).json(); + expect(data.models.some((m) => m.id === "retired-model")).toBe(false); + }); +}); + +describe("criterion 4b — empty catalog → explicit warning", () => { + it("returns warning instead of silent zero", async () => { + stub.mode = "empty"; + const conn = await seedZed("m3"); + const res = await getModels(conn.id); + expect(res.status).toBe(200); + const data = await res.json(); + expect(data.models).toEqual([]); + expect(data.warning).toMatch(/no live models/i); + }); +}); + +describe("criterion 5 — resolver failure → useful warning, no crash", () => { + it("returns 200 with warning text", async () => { + stub.mode = "error"; + const conn = await seedZed("m4"); + const res = await getModels(conn.id); + expect(res.status).toBe(200); + const data = await res.json(); + expect(data.models).toEqual([]); + expect(data.warning).toMatch(/failed to fetch zed models/i); + }); +}); + +describe("criterion 6 (route) — unknown connection → 404", () => { + it("rejects missing connections", async () => { + const res = await getModels("00000000-0000-0000-0000-000000000000"); + expect(res.status).toBe(404); + }); +}); + +describe("criterion 5 (guard) — unsupported provider unchanged", () => { + it("still 400s for providers without a models config", async () => { + const conn = await createProviderConnection({ + provider: "kimchi-nope", + authType: "oauth", + accessToken: "x", + email: `guard-${Date.now()}@example.com`, + testStatus: "active", + }).catch(() => null); + // createProviderConnection may reject unknown providers; either way the + // route must not have gained a zed-shaped branch for others. + if (!conn) return; + const res = await getModels(conn.id); + expect(res.status).toBe(400); + }); +}); diff --git a/tests/unit/zed-native-auth.test.js b/tests/unit/zed-native-auth.test.js new file mode 100644 index 00000000..66acc9cf --- /dev/null +++ b/tests/unit/zed-native-auth.test.js @@ -0,0 +1,266 @@ +// Acceptance suite for the Zed native-app auth fix. +// RUN WITH AN ISOLATED DB: DATA_DIR=$(mktemp -d) npx vitest run unit/zed-native-auth.test.js +// +// Covers criteria: +// 1. Zed proxy starts +// 2. Stray callback (no params) MUST NOT kill session / stop proxy +// 3. Real callback (user_id + access_token) MUST complete session + save connection +// 4. RSA decrypt works (round-trip) +// 5. systemId identical authorize → exchange → stored connection +// 6. register-session failure is distinguishable (backend contract) +// 8. (backend) reopen/re-register creates a fresh session +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import crypto from "node:crypto"; +import { + createZedNativeAuthData, + parseZedCallbackPayload, + decryptZedAccessToken, +} from "open-sse/shared/zedAuth.js"; +import { + startZedProxy, + stopZedProxy, + registerZedSession, + getZedSessionStatus, + clearZedSession, +} from "@/lib/oauth/utils/server.js"; +import { + generateAuthData, + exchangeTokens, +} from "@/lib/oauth/providers/index.js"; + +const realFetch = globalThis.fetch; + +// Never hit the real network in tests: cloud.zed.dev calls are best-effort +// (postExchange try/catch) — fail them fast and loud instead. +beforeEach(() => { + globalThis.fetch = async (url, init) => { + if (String(url).includes("cloud.zed.dev")) { + return new Response("test-stubbed", { status: 500 }); + } + return realFetch(url, init); + }; +}); +afterEach(async () => { + globalThis.fetch = realFetch; + stopZedProxy(); + vi.restoreAllMocks(); +}); + +async function startTestProxy() { + const started = await startZedProxy(0); // random loopback port — parallel-safe + expect(started.success).toBe(true); + return started; +} + +/** Simulate zed.dev: RSA-encrypt a plaintext token with the flow's public key. */ +function encryptForCallback(publicKeyB64Url, plaintext) { + const der = Buffer.from(String(publicKeyB64Url), "base64url"); + const key = crypto.createPublicKey({ key: der, format: "der", type: "pkcs1" }); + return crypto + .publicEncrypt( + { key, padding: crypto.constants.RSA_PKCS1_OAEP_PADDING, oaepHash: "sha256" }, + Buffer.from(plaintext, "utf8"), + ) + .toString("base64url"); +} + +describe("criterion 1 — Zed proxy starts", () => { + it("binds 127.0.0.1 and reports a usable callback URL", async () => { + const started = await startTestProxy(); + expect(started.port).toBeGreaterThan(0); + expect(started.callbackUrl).toBe(`http://127.0.0.1:${started.port}/`); + }); +}); + +describe("criterion 4 — RSA decrypt works", () => { + it("round-trips OAEP-SHA256 through the verifier slot", async () => { + const auth = createZedNativeAuthData({}, { nativeAppPort: 1 }); + const encrypted = encryptForCallback(auth.publicKey, "plaintext-token-abc"); + expect(decryptZedAccessToken(encrypted, auth.privateKeyVerifier)).toBe( + "plaintext-token-abc", + ); + }); + + it("rejects a missing verifier instead of silently failing", () => { + const auth = createZedNativeAuthData({}, { nativeAppPort: 1 }); + const encrypted = encryptForCallback(auth.publicKey, "x"); + expect(() => decryptZedAccessToken(encrypted, null)).toThrow( + /private key verifier/i, + ); + }); + + it("parser keeps strict validation (no weakened acceptance)", () => { + expect(() => parseZedCallbackPayload("")).toThrow(); + expect(() => parseZedCallbackPayload("http://127.0.0.1:1/")).toThrow( + /user_id and access_token/, + ); + expect(() => + parseZedCallbackPayload("http://127.0.0.1:1/?user_id=only-user"), + ).toThrow(/user_id and access_token/); + }); +}); + +describe("criterion 2 — stray callback MUST NOT kill session", () => { + it("bare GET / leaves session pending and proxy listening", async () => { + const started = await startTestProxy(); + const auth = createZedNativeAuthData({}, { nativeAppPort: started.port }); + expect( + registerZedSession({ state: "stray-state-1", codeVerifier: auth.privateKeyVerifier }), + ).toBe(true); + + const res = await realFetch(`http://127.0.0.1:${started.port}/`); + expect(res.status).toBe(200); + + // Session must still be pending (not poisoned to error)… + const session = getZedSessionStatus("stray-state-1"); + expect(session).not.toBeNull(); + expect(session.status).toBe("pending"); + + // …and the SAME server must still own the port (no silent restart). + const again = await startZedProxy(0); + expect(again.port).toBe(started.port); + + clearZedSession("stray-state-1"); + }); + + it("GET /callback with unrelated params leaves session pending", async () => { + const started = await startTestProxy(); + const auth = createZedNativeAuthData({}, { nativeAppPort: started.port }); + registerZedSession({ state: "stray-state-2", codeVerifier: auth.privateKeyVerifier }); + + const res = await realFetch(`http://127.0.0.1:${started.port}/callback?foo=bar`); + expect(res.status).toBe(200); + + const session = getZedSessionStatus("stray-state-2"); + expect(session).not.toBeNull(); + expect(session.status).toBe("pending"); + clearZedSession("stray-state-2"); + }); +}); + +describe("criterion 3 — real callback completes session + saves connection", () => { + it("user_id + access_token → done, decrypted token persisted", async () => { + const started = await startTestProxy(); + const auth = createZedNativeAuthData({}, { nativeAppPort: started.port }); + const state = `real-state-${Date.now()}`; + registerZedSession({ state, codeVerifier: auth.privateKeyVerifier, systemId: auth.systemId }); + + const encrypted = encryptForCallback(auth.publicKey, "decrypted-token-xyz"); + const cb = new URL(`http://127.0.0.1:${started.port}/`); + cb.searchParams.set("user_id", "user-123"); + cb.searchParams.set("access_token", encrypted); + const res = await realFetch(cb.toString()); + expect(res.status).toBe(200); + + const session = getZedSessionStatus(state); + expect(session).not.toBeNull(); + expect(session.status).toBe("done"); + expect(session.connectionId).toBeTruthy(); + + const { getProviderConnectionById } = await import("@/models/index.js"); + const conn = await getProviderConnectionById(session.connectionId); + expect(conn).toBeTruthy(); + expect(conn.provider).toBe("zed"); + expect(conn.accessToken).toBe("decrypted-token-xyz"); + expect(conn.providerSpecificData?.userId).toBe("user-123"); + expect(conn.providerSpecificData?.systemId).toBe(auth.systemId); + + // Proxy stopped itself after the terminal outcome (no orphan listener). + const again = await startZedProxy(0); + expect(again.port).not.toBe(started.port); + stopZedProxy(); + }); +}); + +describe("criterion 5 — systemId stable authorize → exchange → stored", () => { + it("generateAuthData exposes the systemId sent to zed.dev", async () => { + const auth = await generateAuthData("zed", "http://127.0.0.1:59999/", { + nativeAppPort: 59999, + }); + const url = new URL(auth.authUrl); + expect(url.searchParams.get("native_app_port")).toBe("59999"); + // The system_id embedded in the sign-in URL must be observable downstream. + expect(auth.systemId).toBe(url.searchParams.get("system_id")); + expect(auth.systemId).toBeTruthy(); + }); + + it("exchange preserves the registered systemId (no regeneration)", async () => { + const auth = await generateAuthData("zed", "http://127.0.0.1:59998/", { + nativeAppPort: 59998, + }); + // Public key always rides in the authorize URL (mirrors the real flow). + const pubFromUrl = new URL(auth.authUrl).searchParams.get("native_app_public_key"); + expect(pubFromUrl).toBeTruthy(); + const enc2 = encryptForCallback(pubFromUrl, "tok2"); + const tokens = await exchangeTokens( + "zed", + `/?user_id=u1&access_token=${encodeURIComponent(enc2)}`, + null, + auth.codeVerifier, + auth.state, + { systemId: auth.systemId }, + ); + expect(tokens.providerSpecificData.systemId).toBe(auth.systemId); + }); +}); + +describe("criterion 6 — register-session failure is distinguishable", () => { + it("route reports { success: false } when the verifier is missing", async () => { + const { POST } = await import("@/app/api/oauth/[provider]/[action]/route.js"); + const req = new Request("http://localhost/api/oauth/zed/register-session", { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ state: "no-verifier-state" }), + }); + const res = await POST(req, { + params: Promise.resolve({ provider: "zed", action: "register-session" }), + }); + const data = await res.json(); + // Backend contract: failure must be explicit (modal is required to check it). + expect(data.success).toBe(false); + }); +}); + +describe("criterion 8 (backend) — re-register creates a fresh session", () => { + it("a new register supersedes the old state cleanly", async () => { + const a = createZedNativeAuthData({}, { nativeAppPort: 1 }); + const b = createZedNativeAuthData({}, { nativeAppPort: 1 }); + registerZedSession({ state: "old-state", codeVerifier: a.privateKeyVerifier }); + registerZedSession({ state: "new-state", codeVerifier: b.privateKeyVerifier }); + + expect(getZedSessionStatus("old-state")).toBeNull(); + const fresh = getZedSessionStatus("new-state"); + expect(fresh).not.toBeNull(); + expect(fresh.status).toBe("pending"); + expect(fresh.codeVerifier).toBe(b.privateKeyVerifier); + clearZedSession("new-state"); + }); +}); + +describe("criterion L — decrypt failure errors the session but keeps the server", () => { + it("wrong-key token → session error, listener survives for the live attempt", async () => { + const started = await startTestProxy(); + const live = createZedNativeAuthData({}, { nativeAppPort: started.port }); + const other = createZedNativeAuthData({}, { nativeAppPort: started.port }); + const state = `wrongkey-state-${Date.now()}`; + registerZedSession({ state, codeVerifier: live.privateKeyVerifier }); + + // Token encrypted for a DIFFERENT keypair (e.g. superseded popup). + const bad = encryptForCallback(other.publicKey, "not-for-this-key"); + const cb = new URL(`http://127.0.0.1:${started.port}/`); + cb.searchParams.set("user_id", "user-123"); + cb.searchParams.set("access_token", bad); + const res = await realFetch(cb.toString()); + expect(res.status).toBe(200); + + const session = getZedSessionStatus(state); + expect(session).not.toBeNull(); + expect(session.status).toBe("error"); + expect(session.error).toMatch(/decrypt/i); + + // Server must still be alive (same port) for the live attempt. + const again = await startZedProxy(0); + expect(again.port).toBe(started.port); + clearZedSession(state); + }); +});