diff --git a/services/discord-gateway/src/modules/voice-recording/voiceTranscriber.ts b/services/discord-gateway/src/modules/voice-recording/voiceTranscriber.ts index 433f7fd2..5d058538 100644 --- a/services/discord-gateway/src/modules/voice-recording/voiceTranscriber.ts +++ b/services/discord-gateway/src/modules/voice-recording/voiceTranscriber.ts @@ -30,12 +30,16 @@ export async function transcribeRecording( const response = await openai.audio.transcriptions.create({ file: createReadStream(oggPath), - model: "whisper-1", + model: config.AI_VOICE_TRANSCRIPTION_MODEL, // No `language` param → Whisper auto-detects (handles id/en mixed). - response_format: "text", + // `json` (not `text`): 9router/omniroute proxies only json/verbose_json + // transcription responses; `text` returns 400 through the router. + response_format: "json", }); - const text = typeof response === "string" ? response.trim() : null; + // With response_format "json", the SDK returns an object with `text`. + const text = + (typeof response === "string" ? response : response.text)?.trim() || null; if (!text) { logger.warn({ oggPath }, "Empty transcription result"); return null; diff --git a/services/discord-gateway/src/shared/config/index.ts b/services/discord-gateway/src/shared/config/index.ts index 0e92e357..74eac44f 100644 --- a/services/discord-gateway/src/shared/config/index.ts +++ b/services/discord-gateway/src/shared/config/index.ts @@ -340,6 +340,11 @@ export const configSchema = z .optional() .transform((v) => v === "true") .default(false), + // Whisper model routed through the LLM base URL. Through 9router/omniroute + // use a provider-qualified id that has credentials (e.g. + // openrouter/openai/whisper-1) — bare `whisper-1` maps to the `openai` + // provider which the router rejects with "No credentials for provider". + AI_VOICE_TRANSCRIPTION_MODEL: z.string().default("whisper-1"), // ── Auto Delete ───────────────────────────────────────────────────── AUTO_DELETE_FLAGGED_ENABLED: z