Files
9router/open-sse/translator/concerns/usage.js
T
LLL 1f10f9e5c4 fix(qoder): report usage to all clients and stop inlining large attachments
- Coalesce Qoder's empty finish-in-delta frame with the later choices:[] usage
  frame so OpenAI and Claude clients receive prompt_tokens, completion_tokens
  and cache-hit tokens (the dashboard already saw them)
- Upload inlined images through /api/v2/image/upload like qodercli, and stub
  oversized non-image files instead of stuffing 30MB+ data URIs into
  agent_chat_generation
- Emit response.completed -> response.usage for chat-native upstreams so
  /v1/responses clients (Codex CLI, sub2api) no longer log 0/0/0
- Keep Claude message_delta.usage working when usage arrives without choices[0]
- Escalate to the smallest advertised Qoder context tier (200K/400K/1M) when
  the estimated prompt no longer fits max_input_tokens
- Pass apiKey for PAT connections and list hidden enable:false catalog keys
  from /v1/models
2026-09-10 22:08:19 +07:00

101 lines
4.8 KiB
JavaScript

// Build OpenAI usage object. Caller computes prompt/completion/total (provider math).
// Optional details added only when > 0 (matches existing claude/gemini/codex behavior).
export function buildUsage({ promptTokens, completionTokens, totalTokens, cachedTokens = 0, cacheCreationTokens = 0, reasoningTokens = 0 }) {
const usage = { prompt_tokens: promptTokens, completion_tokens: completionTokens, total_tokens: totalTokens };
if (cachedTokens > 0 || cacheCreationTokens > 0) {
usage.prompt_tokens_details = {};
if (cachedTokens > 0) usage.prompt_tokens_details.cached_tokens = cachedTokens;
if (cacheCreationTokens > 0) usage.prompt_tokens_details.cache_creation_tokens = cacheCreationTokens;
}
if (reasoningTokens > 0) {
usage.completion_tokens_details = { reasoning_tokens: reasoningTokens };
}
return usage;
}
const n = (v) => (typeof v === "number" ? v : 0);
// Per-provider raw token field-map + math. Returns buildUsage() args (NOT the usage object).
// Keeps each provider's exact semantics: claude/gemini fold cache+reasoning, others don't.
const USAGE_EXTRACTORS = {
claude(raw) {
const input = n(raw.input_tokens), output = n(raw.output_tokens);
const cacheRead = n(raw.cache_read_input_tokens), cacheCreate = n(raw.cache_creation_input_tokens);
const prompt = input + cacheRead + cacheCreate;
return { promptTokens: prompt, completionTokens: output, totalTokens: prompt + output, cachedTokens: cacheRead, cacheCreationTokens: cacheCreate };
},
gemini(raw) {
const cached = n(raw.cachedContentTokenCount);
const prompt = n(raw.promptTokenCount);
const thoughts = n(raw.thoughtsTokenCount);
const total = n(raw.totalTokenCount);
let candidates = n(raw.candidatesTokenCount);
// Fallback: derive candidates from total when upstream omits it
if (candidates === 0 && total > 0) {
candidates = total - prompt - thoughts;
if (candidates < 0) candidates = 0;
}
return { promptTokens: prompt, completionTokens: candidates + thoughts, totalTokens: total, cachedTokens: cached, reasoningTokens: thoughts };
},
kiro(raw) {
const input = n(raw.inputTokens), output = n(raw.outputTokens);
// ponytail: Amazon Q (Kiro upstream) does not expose cache fields today,
// but pass through any cache_read/cache_creation/cached_tokens if the
// event shape grows them later so cost tracking keeps working without
// a second pass.
const cached = n(raw.cache_read_input_tokens) || n(raw.cachedTokens) || n(raw.cached_tokens);
const cacheCreation = n(raw.cache_creation_input_tokens);
const out = { promptTokens: input, completionTokens: output, totalTokens: input + output };
if (cached > 0) out.cachedTokens = cached;
if (cacheCreation > 0) out.cacheCreationTokens = cacheCreation;
return out;
},
ollama(raw) {
const input = n(raw.prompt_eval_count), output = n(raw.eval_count);
return { promptTokens: input, completionTokens: output, totalTokens: input + output };
},
commandcode(raw) {
const input = n(raw.inputTokens), output = n(raw.outputTokens);
const total = typeof raw.totalTokens === "number" ? raw.totalTokens : input + output;
return { promptTokens: input, completionTokens: output, totalTokens: total };
},
};
// Convert provider-native usage object → OpenAI usage. Returns null if no extractor/raw.
export function toOpenAIUsage(raw, kind) {
const extract = USAGE_EXTRACTORS[kind];
if (!extract || !raw || typeof raw !== "object") return null;
return buildUsage(extract(raw));
}
// Convert an OpenAI-shaped (or already-canonical / Claude-shaped) usage object into the
// Responses API shape emitted by `response.completed`. Details objects are always present
// (like the real API) so proxies that read `input_tokens_details.cached_tokens` never see undefined.
// Returns null when there is nothing countable.
export function toResponsesUsage(usage) {
if (!usage || typeof usage !== "object") return null;
const input = n(usage.prompt_tokens ?? usage.input_tokens);
const output = n(usage.completion_tokens ?? usage.output_tokens);
if (input === 0 && output === 0) return null;
const cached = n(
usage.input_tokens_details?.cached_tokens ??
usage.prompt_tokens_details?.cached_tokens ??
usage.cached_tokens ??
usage.cache_read_input_tokens
);
const reasoning = n(
usage.output_tokens_details?.reasoning_tokens ??
usage.completion_tokens_details?.reasoning_tokens ??
usage.reasoning_tokens
);
const out = {
input_tokens: input,
output_tokens: output,
total_tokens: typeof usage.total_tokens === "number" ? usage.total_tokens : input + output,
input_tokens_details: { cached_tokens: cached },
output_tokens_details: { reasoning_tokens: reasoning },
};
if (usage.estimated) out.estimated = true;
return out;
}