Update version to 0.4.9, enhance README with Trendshift badge, and add new embedding models to providerModels.js. Refactor TTS handling to support additional providers and improve API key validation for media providers.
This commit is contained in:
@@ -10,6 +10,8 @@
|
||||
[](https://www.npmjs.com/package/9router)
|
||||
[](https://www.npmjs.com/package/9router)
|
||||
[](https://github.com/decolua/9router/blob/main/LICENSE)
|
||||
|
||||
<a href="https://trendshift.io/repositories/22628" target="_blank"><img src="https://trendshift.io/api/badge/repositories/22628" alt="decolua%2F9router | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
||||
|
||||
[🚀 Quick Start](#-quick-start) • [💡 Features](#-key-features) • [📖 Setup](#-setup-guide) • [🌐 Website](https://9router.com)
|
||||
|
||||
|
||||
@@ -105,6 +105,9 @@ export const PROVIDER_MODELS = {
|
||||
{ id: "grok-code-fast-1", name: "Grok Code Fast 1" },
|
||||
{ id: "oswe-vscode-prime", name: "Raptor Mini" },
|
||||
{ id: "goldeneye-free-auto", name: "GoldenEye" },
|
||||
// GitHub Copilot - Embedding models
|
||||
{ id: "text-embedding-3-small", name: "Text Embedding 3 Small (GitHub)", type: "embedding" },
|
||||
{ id: "text-embedding-3-large", name: "Text Embedding 3 Large (GitHub)", type: "embedding" },
|
||||
],
|
||||
kr: [ // Kiro AI
|
||||
// { id: "claude-opus-4.5", name: "Claude Opus 4.5" },
|
||||
@@ -378,6 +381,7 @@ export const PROVIDER_MODELS = {
|
||||
{ id: "mistral-large-latest", name: "Mistral Large 3" },
|
||||
{ id: "codestral-latest", name: "Codestral" },
|
||||
{ id: "mistral-medium-latest", name: "Mistral Medium 3" },
|
||||
{ id: "mistral-embed", name: "Mistral Embed", type: "embedding" },
|
||||
],
|
||||
perplexity: [
|
||||
{ id: "sonar-pro", name: "Sonar Pro" },
|
||||
@@ -388,11 +392,14 @@ export const PROVIDER_MODELS = {
|
||||
{ id: "deepseek-ai/DeepSeek-R1", name: "DeepSeek R1" },
|
||||
{ id: "Qwen/Qwen3-235B-A22B", name: "Qwen3 235B" },
|
||||
{ id: "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", name: "Llama 4 Maverick" },
|
||||
{ id: "BAAI/bge-large-en-v1.5", name: "BGE Large EN v1.5", type: "embedding" },
|
||||
{ id: "togethercomputer/m2-bert-80M-8k-retrieval", name: "M2 BERT 80M 8K", type: "embedding" },
|
||||
],
|
||||
fireworks: [
|
||||
{ id: "accounts/fireworks/models/deepseek-v3p1", name: "DeepSeek V3.1" },
|
||||
{ id: "accounts/fireworks/models/llama-v3p3-70b-instruct", name: "Llama 3.3 70B" },
|
||||
{ id: "accounts/fireworks/models/qwen3-235b-a22b", name: "Qwen3 235B" },
|
||||
{ id: "nomic-ai/nomic-embed-text-v1.5", name: "Nomic Embed Text v1.5", type: "embedding" },
|
||||
],
|
||||
cerebras: [
|
||||
{ id: "gpt-oss-120b", name: "GPT OSS 120B" },
|
||||
@@ -410,9 +417,20 @@ export const PROVIDER_MODELS = {
|
||||
nvidia: [
|
||||
{ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5" },
|
||||
{ id: "z-ai/glm4.7", name: "GLM 4.7" },
|
||||
{ id: "nvidia/nv-embedqa-e5-v5", name: "NV EmbedQA E5 v5", type: "embedding" },
|
||||
],
|
||||
nebius: [
|
||||
{ id: "meta-llama/Llama-3.3-70B-Instruct", name: "Llama 3.3 70B Instruct" },
|
||||
{ id: "Qwen/Qwen3-Embedding-8B", name: "Qwen3 Embedding 8B", type: "embedding" },
|
||||
],
|
||||
"voyage-ai": [
|
||||
{ id: "voyage-3-large", name: "Voyage 3 Large", type: "embedding" },
|
||||
{ id: "voyage-3.5", name: "Voyage 3.5", type: "embedding" },
|
||||
{ id: "voyage-3.5-lite", name: "Voyage 3.5 Lite", type: "embedding" },
|
||||
{ id: "voyage-code-3", name: "Voyage Code 3", type: "embedding" },
|
||||
{ id: "voyage-finance-2", name: "Voyage Finance 2", type: "embedding" },
|
||||
{ id: "voyage-law-2", name: "Voyage Law 2", type: "embedding" },
|
||||
{ id: "voyage-multilingual-2", name: "Voyage Multilingual 2", type: "embedding" },
|
||||
],
|
||||
siliconflow: [
|
||||
{ id: "deepseek-ai/DeepSeek-V3.2", name: "DeepSeek V3.2" },
|
||||
|
||||
@@ -7,6 +7,19 @@ import { refreshWithRetry } from "../services/tokenRefresh.js";
|
||||
// Google AI (Gemini) provider aliases / identifiers
|
||||
const GEMINI_PROVIDERS = new Set(["gemini", "google_ai_studio"]);
|
||||
|
||||
// Static map: provider id → embeddings endpoint (OpenAI-compatible body format)
|
||||
const EMBEDDING_URLS = {
|
||||
openai: "https://api.openai.com/v1/embeddings",
|
||||
openrouter: "https://openrouter.ai/api/v1/embeddings",
|
||||
mistral: "https://api.mistral.ai/v1/embeddings",
|
||||
"voyage-ai": "https://api.voyageai.com/v1/embeddings",
|
||||
fireworks: "https://api.fireworks.ai/inference/v1/embeddings",
|
||||
together: "https://api.together.xyz/v1/embeddings",
|
||||
nebius: "https://api.tokenfactory.nebius.com/v1/embeddings",
|
||||
github: "https://models.github.ai/inference/embeddings",
|
||||
nvidia: "https://integrate.api.nvidia.com/v1/embeddings",
|
||||
};
|
||||
|
||||
/**
|
||||
* Check whether a provider targets the Google AI (Gemini) embeddings API.
|
||||
* @param {string} provider
|
||||
@@ -77,22 +90,16 @@ function buildEmbeddingsUrl(provider, model, credentials, input) {
|
||||
return `https://generativelanguage.googleapis.com/v1beta/${modelPath}:embedContent?key=${encodeURIComponent(apiKey)}`;
|
||||
}
|
||||
|
||||
switch (provider) {
|
||||
case "openai":
|
||||
return "https://api.openai.com/v1/embeddings";
|
||||
case "openrouter":
|
||||
return "https://openrouter.ai/api/v1/embeddings";
|
||||
default:
|
||||
// openai-compatible & custom-embedding providers: use their baseUrl + /embeddings
|
||||
if (provider?.startsWith?.("openai-compatible-") || provider?.startsWith?.("custom-embedding-")) {
|
||||
const rawBaseUrl = credentials?.providerSpecificData?.baseUrl || "https://api.openai.com/v1";
|
||||
// Defensive: strip trailing slash and accidental /embeddings to avoid double-append
|
||||
const baseUrl = rawBaseUrl.replace(/\/$/, "").replace(/\/embeddings$/, "");
|
||||
return `${baseUrl}/embeddings`;
|
||||
}
|
||||
// For other providers, attempt to use their base URL pattern with /embeddings path
|
||||
return null;
|
||||
if (EMBEDDING_URLS[provider]) return EMBEDDING_URLS[provider];
|
||||
|
||||
// openai-compatible & custom-embedding providers: use their baseUrl + /embeddings
|
||||
if (provider?.startsWith?.("openai-compatible-") || provider?.startsWith?.("custom-embedding-")) {
|
||||
const rawBaseUrl = credentials?.providerSpecificData?.baseUrl || "https://api.openai.com/v1";
|
||||
// Defensive: strip trailing slash and accidental /embeddings to avoid double-append
|
||||
const baseUrl = rawBaseUrl.replace(/\/$/, "").replace(/\/embeddings$/, "");
|
||||
return `${baseUrl}/embeddings`;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
+232
-24
@@ -455,7 +455,209 @@ async function handleOpenAiTts({ model, input, credentials, responseFormat = "mp
|
||||
return createTtsResponse(base64, "mp3", responseFormat);
|
||||
}
|
||||
|
||||
// ── TTS Provider Registry (DRY) ────────────────────────────────
|
||||
// ── Generic TTS Format Handlers (config-driven via ttsConfig.format) ──────
|
||||
// Parse `model` string as "modelId/voiceId" or "modelId" (modelId may contain slashes — match against known list)
|
||||
function parseModelVoice(model, defaultModel = "", defaultVoice = "", knownModels = []) {
|
||||
if (!model) return { modelId: defaultModel, voiceId: defaultVoice };
|
||||
// Find longest known model id that prefixes `model`
|
||||
const known = knownModels.map((m) => m.id || m).filter(Boolean).sort((a, b) => b.length - a.length);
|
||||
for (const id of known) {
|
||||
if (model === id) return { modelId: id, voiceId: defaultVoice };
|
||||
if (model.startsWith(`${id}/`)) return { modelId: id, voiceId: model.slice(id.length + 1) };
|
||||
}
|
||||
// Fallback: split on last "/" so "vendor/model/voice" → model="vendor/model", voice="voice"
|
||||
const idx = model.lastIndexOf("/");
|
||||
if (idx > 0) return { modelId: model.slice(0, idx), voiceId: model.slice(idx + 1) };
|
||||
return { modelId: defaultModel || model, voiceId: defaultVoice || model };
|
||||
}
|
||||
|
||||
// Convert upstream Response (binary audio) to { base64, format }
|
||||
async function responseToBase64(res, defaultFormat = "mp3") {
|
||||
const buf = await res.arrayBuffer();
|
||||
if (buf.byteLength < 100) throw new Error("Upstream returned empty audio");
|
||||
const ctype = res.headers.get("content-type") || "";
|
||||
let format = defaultFormat;
|
||||
if (ctype.includes("wav")) format = "wav";
|
||||
else if (ctype.includes("mpeg") || ctype.includes("mp3")) format = "mp3";
|
||||
else if (ctype.includes("ogg")) format = "ogg";
|
||||
return { base64: Buffer.from(buf).toString("base64"), format };
|
||||
}
|
||||
|
||||
async function throwUpstreamError(res) {
|
||||
const text = await res.text().catch(() => "");
|
||||
let msg = `Upstream error (${res.status})`;
|
||||
try {
|
||||
const parsed = JSON.parse(text);
|
||||
msg = parsed?.error?.message || parsed?.message || parsed?.detail?.message || (typeof parsed?.detail === "string" ? parsed.detail : null) || text || msg;
|
||||
} catch { msg = text || msg; }
|
||||
throw new Error(msg);
|
||||
}
|
||||
|
||||
// Hyperbolic: POST { text } → { audio: base64 }
|
||||
async function ttsHyperbolic({ baseUrl, apiKey, text }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${apiKey}` },
|
||||
body: JSON.stringify({ text }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
const data = await res.json();
|
||||
return { base64: data.audio, format: "mp3" };
|
||||
}
|
||||
|
||||
// Deepgram: model via query, Token auth, returns binary
|
||||
async function ttsDeepgram({ baseUrl, apiKey, text, modelId }) {
|
||||
const url = new URL(baseUrl);
|
||||
url.searchParams.set("model", modelId || "aura-asteria-en");
|
||||
const res = await fetch(url.toString(), {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Token ${apiKey}` },
|
||||
body: JSON.stringify({ text }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Nvidia NIM: POST { input: { text }, voice, model } → binary
|
||||
async function ttsNvidia({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${apiKey}` },
|
||||
body: JSON.stringify({ input: { text }, voice: voiceId || "default", model: modelId }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// HuggingFace: POST {baseUrl}/{modelId} { inputs: text } → binary
|
||||
async function ttsHuggingFace({ baseUrl, apiKey, text, modelId }) {
|
||||
if (!modelId || modelId.includes("..")) throw new Error("Invalid HuggingFace model ID");
|
||||
const res = await fetch(`${baseUrl}/${modelId}`, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Bearer ${apiKey}` },
|
||||
body: JSON.stringify({ inputs: text }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Inworld: POST { text, voiceId, modelId, audioConfig } → JSON { audioContent }
|
||||
async function ttsInworld({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json", "Authorization": `Basic ${apiKey}` },
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
voiceId: voiceId || "Alex",
|
||||
modelId: modelId || "inworld-tts-1.5-mini",
|
||||
audioConfig: { audioEncoding: "MP3" },
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
const data = await res.json();
|
||||
if (!data.audioContent) throw new Error("Inworld TTS returned no audio");
|
||||
return { base64: data.audioContent, format: "mp3" };
|
||||
}
|
||||
|
||||
// Cartesia: POST { model_id, transcript, voice, output_format } → binary
|
||||
async function ttsCartesia({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"X-API-Key": apiKey,
|
||||
"Cartesia-Version": "2024-06-10",
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model_id: modelId || "sonic-2",
|
||||
transcript: text,
|
||||
...(voiceId ? { voice: { mode: "id", id: voiceId } } : {}),
|
||||
output_format: { container: "mp3", bit_rate: 128000, sample_rate: 44100 },
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// PlayHT: token format "userId:apiKey", voice = s3 URL
|
||||
async function ttsPlayHt({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const [userId, key] = (apiKey || ":").split(":");
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"Accept": "audio/mpeg",
|
||||
"X-USER-ID": userId || "",
|
||||
"Authorization": `Bearer ${key || apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
voice: voiceId || "s3://voice-cloning-zero-shot/d9ff78ba-d016-47f6-b0ef-dd630f59414e/female-cs/manifest.json",
|
||||
voice_engine: modelId || "PlayDialog",
|
||||
output_format: "mp3",
|
||||
speed: 1,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Coqui (local, noAuth): POST { text, speaker_id } → WAV
|
||||
async function ttsCoqui({ baseUrl, text, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ text, ...(voiceId ? { speaker_id: voiceId } : {}) }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// Tortoise (local, noAuth): POST { text, voice } → binary
|
||||
async function ttsTortoise({ baseUrl, text, voiceId }) {
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ text, voice: voiceId || "random" }),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "wav");
|
||||
}
|
||||
|
||||
// OpenAI-compatible (qwen3-tts, openai-compat): POST { model, input, voice } → binary
|
||||
async function ttsOpenAiCompat({ baseUrl, apiKey, text, modelId, voiceId }) {
|
||||
const headers = { "Content-Type": "application/json" };
|
||||
if (apiKey) headers["Authorization"] = `Bearer ${apiKey}`;
|
||||
const res = await fetch(baseUrl, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
input: text,
|
||||
voice: voiceId || "alloy",
|
||||
response_format: "mp3",
|
||||
speed: 1.0,
|
||||
}),
|
||||
});
|
||||
if (!res.ok) await throwUpstreamError(res);
|
||||
return responseToBase64(res, "mp3");
|
||||
}
|
||||
|
||||
// Format → handler dispatcher (DRY)
|
||||
const FORMAT_HANDLERS = {
|
||||
hyperbolic: ttsHyperbolic,
|
||||
deepgram: ttsDeepgram,
|
||||
"nvidia-tts": ttsNvidia,
|
||||
"huggingface-tts": ttsHuggingFace,
|
||||
inworld: ttsInworld,
|
||||
cartesia: ttsCartesia,
|
||||
playht: ttsPlayHt,
|
||||
coqui: ttsCoqui,
|
||||
tortoise: ttsTortoise,
|
||||
openai: ttsOpenAiCompat,
|
||||
};
|
||||
|
||||
// ── TTS Provider Registry (legacy noAuth + special providers) ──────────
|
||||
const TTS_PROVIDERS = {
|
||||
"google-tts": {
|
||||
synthesize: async (text, model) => {
|
||||
@@ -480,15 +682,10 @@ const TTS_PROVIDERS = {
|
||||
},
|
||||
"elevenlabs": {
|
||||
synthesize: async (text, model, credentials) => {
|
||||
if (!credentials?.apiKey) {
|
||||
throw new Error("ElevenLabs API key required");
|
||||
}
|
||||
// model format: "voice_id" or "model_id/voice_id"
|
||||
if (!credentials?.apiKey) throw new Error("ElevenLabs API key required");
|
||||
let modelId = "eleven_flash_v2_5";
|
||||
let voiceId = model;
|
||||
if (model && model.includes("/")) {
|
||||
[modelId, voiceId] = model.split("/");
|
||||
}
|
||||
if (model && model.includes("/")) [modelId, voiceId] = model.split("/");
|
||||
const base64 = await elevenlabsTts(text, voiceId, credentials.apiKey, modelId);
|
||||
return { base64, format: "mp3" };
|
||||
},
|
||||
@@ -508,15 +705,24 @@ const TTS_PROVIDERS = {
|
||||
},
|
||||
};
|
||||
|
||||
// ── Generic dispatcher: providers with ttsConfig.format ────────────────
|
||||
// Resolves to TTS_PROVIDERS first; falls back to ttsConfig.format dispatch.
|
||||
async function synthesizeViaConfig(provider, text, model, credentials) {
|
||||
const { AI_PROVIDERS } = await import("@/shared/constants/providers");
|
||||
const cfg = AI_PROVIDERS[provider]?.ttsConfig;
|
||||
if (!cfg) return null;
|
||||
const handler = FORMAT_HANDLERS[cfg.format];
|
||||
if (!handler) return null;
|
||||
const apiKey = credentials?.apiKey;
|
||||
if (cfg.authType !== "none" && !apiKey) throw new Error(`${provider} API key required`);
|
||||
const defaultModel = cfg.models?.[0]?.id || "";
|
||||
const { modelId, voiceId } = parseModelVoice(model, defaultModel, "", cfg.models || []);
|
||||
return handler({ baseUrl: cfg.baseUrl, apiKey, text, modelId, voiceId });
|
||||
}
|
||||
|
||||
// ── Core handler ───────────────────────────────────────────────
|
||||
/**
|
||||
* Synthesize text to audio.
|
||||
* @param {object} options
|
||||
* @param {string} options.provider - "google-tts" | "edge-tts" | "local-device" | "openai"
|
||||
* @param {string} options.model - voice/lang id
|
||||
* @param {string} options.input - text to synthesize
|
||||
* @param {object} [options.credentials] - required for openai
|
||||
* @param {string} [options.responseFormat] - "mp3" (default) | "json" (base64)
|
||||
* @returns {Promise<{success, response, status?, error?}>}
|
||||
*/
|
||||
export async function handleTtsCore({ provider, model, input, credentials, responseFormat = "mp3" }) {
|
||||
@@ -525,18 +731,20 @@ export async function handleTtsCore({ provider, model, input, credentials, respo
|
||||
}
|
||||
|
||||
const ttsProvider = TTS_PROVIDERS[provider];
|
||||
if (!ttsProvider) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Provider '${provider}' does not support TTS via this route.`);
|
||||
}
|
||||
|
||||
try {
|
||||
const result = await ttsProvider.synthesize(input.trim(), model, credentials, responseFormat);
|
||||
|
||||
// OpenAI returns full response object
|
||||
if (result.success !== undefined) return result;
|
||||
|
||||
// Other providers return { base64, format }
|
||||
return createTtsResponse(result.base64, result.format, responseFormat);
|
||||
// Legacy/special providers (google-tts, edge-tts, local-device, elevenlabs, openai, openrouter)
|
||||
if (ttsProvider) {
|
||||
const result = await ttsProvider.synthesize(input.trim(), model, credentials, responseFormat);
|
||||
if (result.success !== undefined) return result;
|
||||
return createTtsResponse(result.base64, result.format, responseFormat);
|
||||
}
|
||||
|
||||
// Generic config-driven dispatcher (hyperbolic, deepgram, nvidia, huggingface, inworld, cartesia, playht, coqui, tortoise, qwen, ...)
|
||||
const result = await synthesizeViaConfig(provider, input.trim(), model, credentials);
|
||||
if (result) return createTtsResponse(result.base64, result.format, responseFormat);
|
||||
|
||||
return createErrorResult(HTTP_STATUS.BAD_REQUEST, `Provider '${provider}' does not support TTS via this route.`);
|
||||
} catch (err) {
|
||||
return createErrorResult(HTTP_STATUS.BAD_GATEWAY, err.message || "TTS synthesis failed");
|
||||
}
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "9router-app",
|
||||
"version": "0.4.8",
|
||||
"version": "0.4.9",
|
||||
"description": "9Router web dashboard",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 16 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 2.5 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 3.4 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 1.2 KiB |
@@ -364,6 +364,8 @@ function TtsExampleCard({ providerId }) {
|
||||
const [countryVoices, setCountryVoices] = useState([]);
|
||||
const [selectedLang, setSelectedLang] = useState("");
|
||||
const [selectedModel, setSelectedModel] = useState(() => {
|
||||
const cfgModels = AI_PROVIDERS[providerId]?.ttsConfig?.models;
|
||||
if (cfgModels?.length) return cfgModels[0].id;
|
||||
if (config.hasModelSelector && config.modelKey) {
|
||||
const models = getModelsByProviderId(config.modelKey);
|
||||
return models?.[0]?.id || "";
|
||||
@@ -430,6 +432,8 @@ function TtsExampleCard({ providerId }) {
|
||||
}
|
||||
}
|
||||
// api-language (edge-tts, local-device, elevenlabs): NO default load, wait for user to pick language
|
||||
// config (nvidia, hyperbolic, deepgram, huggingface, cartesia, playht, coqui, tortoise, inworld, qwen):
|
||||
// use ttsConfig.models for model selector; voice is empty by default (backend uses provider default)
|
||||
}, [providerId]);
|
||||
|
||||
// Update voices when model changes (voicesPerModel providers)
|
||||
@@ -501,11 +505,14 @@ function TtsExampleCard({ providerId }) {
|
||||
: languages;
|
||||
|
||||
const endpoint = useTunnel ? tunnelEndpoint : localEndpoint;
|
||||
// For ElevenLabs: use voiceId (editable) instead of selectedVoice
|
||||
const activeVoiceId = config.hasVoiceIdInput ? voiceId : selectedVoice;
|
||||
const modelFull = config.hasModelSelector && activeVoiceId && selectedModel
|
||||
? `${providerAlias}/${selectedModel}/${activeVoiceId}`
|
||||
: activeVoiceId ? `${providerAlias}/${activeVoiceId}` : "";
|
||||
// For ElevenLabs/config-driven: prefer manual voiceId (if any), else fall back to selectedVoice
|
||||
const activeVoiceId = config.hasVoiceIdInput ? (voiceId || selectedVoice) : selectedVoice;
|
||||
const modelFull = (() => {
|
||||
if (config.hasModelSelector && selectedModel && activeVoiceId) return `${providerAlias}/${selectedModel}/${activeVoiceId}`;
|
||||
if (config.hasModelSelector && selectedModel) return `${providerAlias}/${selectedModel}`;
|
||||
if (activeVoiceId) return `${providerAlias}/${activeVoiceId}`;
|
||||
return "";
|
||||
})();
|
||||
|
||||
const curlSnippet = `curl -X POST ${endpoint}/v1/audio/speech${responseFormat === "json" ? "?response_format=json" : ""} \\
|
||||
-H "Content-Type: application/json" \\
|
||||
@@ -584,15 +591,17 @@ function TtsExampleCard({ providerId }) {
|
||||
</span>
|
||||
</Row>
|
||||
|
||||
{/* Model selector (OpenAI, ElevenLabs) */}
|
||||
{config.hasModelSelector && config.modelKey && (
|
||||
{/* Model selector — prefer ttsConfig.models, else providerModels via modelKey */}
|
||||
{config.hasModelSelector && (config.modelKey || AI_PROVIDERS[providerId]?.ttsConfig?.models?.length) && (
|
||||
<Row label="Model">
|
||||
<select
|
||||
value={selectedModel}
|
||||
onChange={(e) => setSelectedModel(e.target.value)}
|
||||
className="w-full px-3 py-1.5 text-sm border border-border rounded-lg bg-background focus:outline-none focus:border-primary"
|
||||
>
|
||||
{(getModelsByProviderId(config.modelKey) || []).map((m) => (
|
||||
{((AI_PROVIDERS[providerId]?.ttsConfig?.models?.length
|
||||
? AI_PROVIDERS[providerId].ttsConfig.models
|
||||
: getModelsByProviderId(config.modelKey)) || []).map((m) => (
|
||||
<option key={m.id} value={m.id}>{m.name || m.id}</option>
|
||||
))}
|
||||
</select>
|
||||
@@ -1446,13 +1455,14 @@ export default function MediaProviderDetailPage() {
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Provider Info — config-driven, supports searchConfig, fetchConfig, searchViaChat */}
|
||||
{!isCustom && (provider.searchConfig || provider.fetchConfig || provider.searchViaChat) && (
|
||||
{/* Provider Info — config-driven, supports searchConfig, fetchConfig, ttsConfig, embeddingConfig, searchViaChat */}
|
||||
{!isCustom && (provider.searchConfig || provider.fetchConfig || provider.ttsConfig || provider.embeddingConfig || provider.searchViaChat) && (
|
||||
<ProviderInfoCard
|
||||
config={
|
||||
kind === "webFetch"
|
||||
? provider.fetchConfig
|
||||
: provider.searchConfig || { mode: "chat-completions", defaultModel: provider.searchViaChat?.defaultModel, costPerQuery: 0 }
|
||||
kind === "webFetch" ? provider.fetchConfig
|
||||
: kind === "tts" ? provider.ttsConfig
|
||||
: kind === "embedding" ? provider.embeddingConfig
|
||||
: provider.searchConfig || { mode: "chat-completions", defaultModel: provider.searchViaChat?.defaultModel, pricingUrl: provider.searchViaChat?.pricingUrl, freeTier: provider.searchViaChat?.freeTier }
|
||||
}
|
||||
provider={provider}
|
||||
title={`${kindConfig.label} Config`}
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
import { NextResponse } from "next/server";
|
||||
import { getProviderConnections } from "@/lib/localDb";
|
||||
|
||||
const langNames = new Intl.DisplayNames(["en"], { type: "language" });
|
||||
|
||||
/**
|
||||
* GET /api/media-providers/tts/deepgram/voices[?lang=en]
|
||||
* Returns { languages, byLang } grouped by language code (same shape as edge-tts/elevenlabs/inworld)
|
||||
* Each Deepgram voice = one model (canonical_name like "aura-2-thalia-en")
|
||||
*/
|
||||
export async function GET(request) {
|
||||
try {
|
||||
const { searchParams } = new URL(request.url);
|
||||
const langFilter = searchParams.get("lang");
|
||||
|
||||
const connections = await getProviderConnections({ provider: "deepgram", isActive: true });
|
||||
const apiKey = connections[0]?.apiKey;
|
||||
if (!apiKey) return NextResponse.json({ error: "No Deepgram connection found" }, { status: 400 });
|
||||
|
||||
const res = await fetch("https://api.deepgram.com/v1/models", {
|
||||
headers: { "Authorization": `Token ${apiKey}` },
|
||||
});
|
||||
if (!res.ok) {
|
||||
const text = await res.text().catch(() => "");
|
||||
return NextResponse.json({ error: `Deepgram API ${res.status}: ${text || "Failed"}` }, { status: 502 });
|
||||
}
|
||||
const data = await res.json();
|
||||
const ttsModels = data.tts || [];
|
||||
|
||||
const byLang = {};
|
||||
for (const m of ttsModels) {
|
||||
// Deepgram returns `languages: ["en"]` or sometimes language inferred from canonical_name suffix
|
||||
const langs = Array.isArray(m.languages) && m.languages.length
|
||||
? m.languages
|
||||
: [m.canonical_name?.split("-").pop() || "en"];
|
||||
for (const code of langs) {
|
||||
if (!byLang[code]) {
|
||||
byLang[code] = {
|
||||
code,
|
||||
name: (() => { try { return langNames.of(code); } catch { return code; } })(),
|
||||
voices: [],
|
||||
};
|
||||
}
|
||||
const voiceId = m.canonical_name || m.name;
|
||||
if (!byLang[code].voices.find((x) => x.id === voiceId)) {
|
||||
byLang[code].voices.push({
|
||||
id: voiceId,
|
||||
name: m.name || voiceId,
|
||||
gender: m.metadata?.tags?.find((t) => t === "masculine" || t === "feminine") || "",
|
||||
lang: code,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const languages = Object.values(byLang).sort((a, b) => a.name.localeCompare(b.name));
|
||||
|
||||
if (langFilter) {
|
||||
return NextResponse.json({ voices: byLang[langFilter]?.voices || [] });
|
||||
}
|
||||
return NextResponse.json({ languages, byLang });
|
||||
} catch (err) {
|
||||
return NextResponse.json({ error: err.message || "Failed to fetch voices" }, { status: 502 });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
import { NextResponse } from "next/server";
|
||||
import { getProviderConnections } from "@/lib/localDb";
|
||||
|
||||
const langNames = new Intl.DisplayNames(["en"], { type: "language" });
|
||||
|
||||
/**
|
||||
* GET /api/media-providers/tts/inworld/voices[?lang=en]
|
||||
* Returns { languages, byLang } grouped by language code (same shape as edge-tts/elevenlabs)
|
||||
*/
|
||||
export async function GET(request) {
|
||||
try {
|
||||
const { searchParams } = new URL(request.url);
|
||||
const langFilter = searchParams.get("lang");
|
||||
|
||||
const connections = await getProviderConnections({ provider: "inworld", isActive: true });
|
||||
const apiKey = connections[0]?.apiKey;
|
||||
if (!apiKey) return NextResponse.json({ error: "No Inworld connection found" }, { status: 400 });
|
||||
|
||||
const res = await fetch("https://api.inworld.ai/tts/v1/voices", {
|
||||
headers: { "Authorization": `Basic ${apiKey}` },
|
||||
});
|
||||
if (!res.ok) {
|
||||
const text = await res.text().catch(() => "");
|
||||
return NextResponse.json({ error: `Inworld API ${res.status}: ${text || "Failed"}` }, { status: 502 });
|
||||
}
|
||||
const data = await res.json();
|
||||
const voices = data.voices || [];
|
||||
|
||||
const byLang = {};
|
||||
for (const v of voices) {
|
||||
// Each voice has `languages: ["en", "es", ...]`
|
||||
const langs = Array.isArray(v.languages) && v.languages.length ? v.languages : ["en"];
|
||||
for (const code of langs) {
|
||||
if (!byLang[code]) {
|
||||
byLang[code] = {
|
||||
code,
|
||||
name: (() => { try { return langNames.of(code); } catch { return code; } })(),
|
||||
voices: [],
|
||||
};
|
||||
}
|
||||
if (!byLang[code].voices.find((x) => x.id === v.voiceId)) {
|
||||
byLang[code].voices.push({
|
||||
id: v.voiceId,
|
||||
name: v.displayName || v.voiceId,
|
||||
gender: v.gender || "",
|
||||
lang: code,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const languages = Object.values(byLang).sort((a, b) => a.name.localeCompare(b.name));
|
||||
|
||||
if (langFilter) {
|
||||
return NextResponse.json({ voices: byLang[langFilter]?.voices || [] });
|
||||
}
|
||||
return NextResponse.json({ languages, byLang });
|
||||
} catch (err) {
|
||||
return NextResponse.json({ error: err.message || "Failed to fetch voices" }, { status: 502 });
|
||||
}
|
||||
}
|
||||
@@ -40,6 +40,43 @@ async function probeWebProvider(provider, apiKey) {
|
||||
return res.status !== 401 && res.status !== 403;
|
||||
}
|
||||
|
||||
// Probe a tts/embedding provider using ttsConfig/embeddingConfig.
|
||||
// Returns true if API key is accepted (status !== 401 && !== 403); null to skip.
|
||||
async function probeMediaProvider(provider, apiKey) {
|
||||
const p = AI_PROVIDERS[provider];
|
||||
if (!p) return null;
|
||||
// Only probe providers that are media-only (not LLM dual-purpose, let LLM validate handle those)
|
||||
const kinds = p.serviceKinds || ["llm"];
|
||||
const isMediaOnly = kinds.every((k) => k === "tts" || k === "embedding" || k === "stt");
|
||||
if (!isMediaOnly) return null;
|
||||
const cfg = p.ttsConfig || p.embeddingConfig;
|
||||
if (!cfg) return null;
|
||||
if (p.noAuth || cfg.authType === "none") return true;
|
||||
// Skip auth schemes that need provider-specific data
|
||||
if (cfg.authHeader === "playht" || cfg.authHeader === "aws-sigv4") return null;
|
||||
|
||||
const headers = { "Content-Type": "application/json" };
|
||||
|
||||
// Apply auth based on authHeader
|
||||
switch (cfg.authHeader) {
|
||||
case "bearer": headers["Authorization"] = `Bearer ${apiKey}`; break;
|
||||
case "x-api-key": headers["x-api-key"] = apiKey; break;
|
||||
case "xi-api-key": headers["xi-api-key"] = apiKey; break;
|
||||
case "token": headers["Authorization"] = `Token ${apiKey}`; break;
|
||||
case "basic": headers["Authorization"] = `Basic ${apiKey}`; break;
|
||||
default: return null;
|
||||
}
|
||||
|
||||
// Minimal POST body — server will reject auth before validating body
|
||||
const res = await fetch(cfg.baseUrl, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({ input: "ping", text: "ping", model: cfg.models?.[0]?.id || "test" }),
|
||||
signal: AbortSignal.timeout(8000),
|
||||
});
|
||||
return res.status !== 401 && res.status !== 403;
|
||||
}
|
||||
|
||||
// POST /api/providers/validate - Validate API key with provider
|
||||
export async function POST(request) {
|
||||
try {
|
||||
@@ -192,6 +229,15 @@ export async function POST(request) {
|
||||
});
|
||||
}
|
||||
|
||||
// Generic probe for tts/embedding providers (config-driven)
|
||||
const mediaResult = await probeMediaProvider(provider, apiKey);
|
||||
if (mediaResult !== null) {
|
||||
return NextResponse.json({
|
||||
valid: mediaResult,
|
||||
error: mediaResult ? null : "Invalid API key",
|
||||
});
|
||||
}
|
||||
|
||||
switch (provider) {
|
||||
case "openai":
|
||||
const openaiRes = await fetch("https://api.openai.com/v1/models", {
|
||||
|
||||
+6
-1
@@ -15,6 +15,11 @@ const URL_PATTERNS = {
|
||||
cursor: ["/BidiAppend", "/RunSSE", "/RunPoll", "/Run"],
|
||||
};
|
||||
|
||||
// Synonym map: rawModel from request → canonical alias key in mitmAlias DB
|
||||
const MODEL_SYNONYMS = {
|
||||
antigravity: { "gemini-default": "gemini-3-flash" },
|
||||
};
|
||||
|
||||
function getToolForHost(host) {
|
||||
const h = (host || "").split(":")[0];
|
||||
if (h === "api.individual.githubcopilot.com") return "copilot";
|
||||
@@ -24,4 +29,4 @@ function getToolForHost(host) {
|
||||
return null;
|
||||
}
|
||||
|
||||
module.exports = { TARGET_HOSTS, URL_PATTERNS, getToolForHost };
|
||||
module.exports = { TARGET_HOSTS, URL_PATTERNS, MODEL_SYNONYMS, getToolForHost };
|
||||
|
||||
+6
-4
@@ -5,14 +5,14 @@ const dns = require("dns");
|
||||
const { promisify } = require("util");
|
||||
const { execSync } = require("child_process");
|
||||
const { log, err } = require("./logger");
|
||||
const { TARGET_HOSTS, URL_PATTERNS, getToolForHost } = require("./config");
|
||||
const { TARGET_HOSTS, URL_PATTERNS, MODEL_SYNONYMS, getToolForHost } = require("./config");
|
||||
const { DATA_DIR, MITM_DIR } = require("./paths");
|
||||
const { getCertForDomain } = require("./cert/generate");
|
||||
|
||||
const DB_FILE = path.join(DATA_DIR, "db.json");
|
||||
const LOCAL_PORT = 443;
|
||||
const IS_WIN = process.platform === "win32";
|
||||
const ENABLE_FILE_LOG = false;
|
||||
const ENABLE_FILE_LOG = true;
|
||||
const LOG_DIR = path.join(DATA_DIR, "logs", "mitm");
|
||||
const INTERNAL_REQUEST_HEADER = { name: "x-request-source", value: "local" };
|
||||
|
||||
@@ -107,9 +107,11 @@ function getMappedModel(tool, model) {
|
||||
const db = JSON.parse(fs.readFileSync(DB_FILE, "utf-8"));
|
||||
const aliases = db.mitmAlias?.[tool];
|
||||
if (!aliases) return null;
|
||||
if (aliases[model]) return aliases[model];
|
||||
// Normalize via synonym map (e.g., gemini-default → gemini-3-flash)
|
||||
const lookup = MODEL_SYNONYMS?.[tool]?.[model] || model;
|
||||
if (aliases[lookup]) return aliases[lookup];
|
||||
// Prefix match fallback
|
||||
const prefixKey = Object.keys(aliases).find(k => k && aliases[k] && (model.startsWith(k) || k.startsWith(model)));
|
||||
const prefixKey = Object.keys(aliases).find(k => k && aliases[k] && (lookup.startsWith(k) || k.startsWith(lookup)));
|
||||
return prefixKey ? aliases[prefixKey] : null;
|
||||
} catch { return null; }
|
||||
}
|
||||
|
||||
@@ -8,6 +8,8 @@ const FIELD_SCHEMA = {
|
||||
defaultModel: { label: "Model", format: (v) => v, mono: true },
|
||||
baseUrl: { label: "Endpoint", format: (v) => v, isLink: true, mono: true },
|
||||
costPerQuery: { label: "Cost / call", format: (v) => v === 0 ? "Free" : `$${v.toFixed(4)}` },
|
||||
pricingUrl: { label: "Pricing", format: () => "View pricing", isLink: true },
|
||||
freeTier: { label: "Free tier", format: (v) => v },
|
||||
freeMonthlyQuota: { label: "Free quota", format: (v) => v === 0 ? "—" : v >= 999999 ? "Unlimited" : `${v.toLocaleString()} / mo` },
|
||||
searchTypes: { label: "Types", format: (v) => v.join(", ") },
|
||||
formats: { label: "Formats", format: (v) => v.join(", ") },
|
||||
@@ -30,6 +32,7 @@ export default function ProviderInfoCard({ config, provider, title = "Provider I
|
||||
}));
|
||||
|
||||
const signupUrl = provider?.notice?.apiKeyUrl || provider?.website;
|
||||
const noticeText = provider?.notice?.text;
|
||||
|
||||
return (
|
||||
<Card>
|
||||
@@ -67,6 +70,12 @@ export default function ProviderInfoCard({ config, provider, title = "Provider I
|
||||
)}
|
||||
</div>
|
||||
))}
|
||||
{noticeText && (
|
||||
<div className="flex items-start gap-3 min-w-0 sm:col-span-2">
|
||||
<span className="text-xs text-text-muted w-28 shrink-0 mt-0.5">Notice</span>
|
||||
<span className="text-sm text-text-main leading-relaxed">{noticeText}</span>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</Card>
|
||||
);
|
||||
|
||||
@@ -12,7 +12,7 @@ export const MITM_TOOLS = {
|
||||
defaultModels: [
|
||||
{ id: "gemini-3.1-pro-high", name: "Gemini 3.1 Pro High", alias: "gemini-3.1-pro-high" },
|
||||
{ id: "gemini-3.1-pro-low", name: "Gemini 3.1 Pro Low", alias: "gemini-3.1-pro-low" },
|
||||
{ id: "gemini-3-flash", name: "Gemini 3 Flash", alias: "gemini-3-flash" },
|
||||
{ id: "gemini-3-flash", name: "Gemini 3 Flash / Default", alias: "gemini-3-flash" },
|
||||
{ id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6", alias: "claude-sonnet-4-6" },
|
||||
{ id: "claude-opus-4-6-thinking", name: "Claude Opus 4.6 Thinking", alias: "claude-opus-4-6-thinking" },
|
||||
{ id: "gpt-oss-120b-medium", name: "GPT OSS 120B Medium", alias: "gpt-oss-120b-medium" },
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -48,4 +48,65 @@ export const TTS_PROVIDER_CONFIG = {
|
||||
hasBrowseButton: true,
|
||||
voiceSource: "api-language", // from API with language picker
|
||||
},
|
||||
// ── Config-driven providers (load models from providers.js → ttsConfig.models) ──
|
||||
"nvidia": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
hasVoiceIdInput: true,
|
||||
voiceSource: "config",
|
||||
},
|
||||
"hyperbolic": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
voiceSource: "config",
|
||||
},
|
||||
"deepgram": {
|
||||
hasModelSelector: false,
|
||||
hasBrowseButton: true,
|
||||
voiceSource: "api-language",
|
||||
apiEndpoint: "/api/media-providers/tts/deepgram/voices",
|
||||
},
|
||||
"huggingface": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
voiceSource: "config",
|
||||
},
|
||||
"cartesia": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
hasVoiceIdInput: true,
|
||||
voiceSource: "config",
|
||||
},
|
||||
"playht": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
hasVoiceIdInput: true,
|
||||
voiceSource: "config",
|
||||
},
|
||||
"coqui": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
hasVoiceIdInput: true,
|
||||
voiceSource: "config",
|
||||
},
|
||||
"tortoise": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
hasVoiceIdInput: true,
|
||||
voiceSource: "config",
|
||||
},
|
||||
"inworld": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: true,
|
||||
hasVoiceIdInput: true,
|
||||
voiceSource: "api-language",
|
||||
modelKey: "inworld-tts-models",
|
||||
apiEndpoint: "/api/media-providers/tts/inworld/voices",
|
||||
},
|
||||
"qwen": {
|
||||
hasModelSelector: true,
|
||||
hasBrowseButton: false,
|
||||
hasVoiceIdInput: true,
|
||||
voiceSource: "config",
|
||||
},
|
||||
};
|
||||
|
||||
@@ -7,10 +7,15 @@ import { getModelInfo } from "../services/model.js";
|
||||
import { handleTtsCore } from "open-sse/handlers/ttsCore.js";
|
||||
import { errorResponse, unavailableResponse } from "open-sse/utils/error.js";
|
||||
import { HTTP_STATUS } from "open-sse/config/runtimeConfig.js";
|
||||
import { AI_PROVIDERS } from "@/shared/constants/providers";
|
||||
import * as log from "../utils/logger.js";
|
||||
|
||||
// Providers that require stored credentials (not noAuth)
|
||||
const CREDENTIALED_PROVIDERS = new Set(["openai", "elevenlabs", "openrouter"]);
|
||||
// Derived from providers.js: any TTS provider not noAuth requires stored credentials
|
||||
const CREDENTIALED_PROVIDERS = new Set(
|
||||
Object.entries(AI_PROVIDERS)
|
||||
.filter(([, p]) => p.serviceKinds?.includes("tts") && !p.noAuth && p.ttsConfig?.authType !== "none")
|
||||
.map(([id]) => id)
|
||||
);
|
||||
|
||||
export async function handleTts(request) {
|
||||
let body;
|
||||
|
||||
Reference in New Issue
Block a user