From ef7708bf7d0f47aabde92e4d2d04351858690199 Mon Sep 17 00:00:00 2001 From: asepharyana Date: Thu, 24 Sep 2026 16:05:36 +0700 Subject: [PATCH] feat(gmw): route all LLM traffic through 9router MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GMW moves off omniroute (100.121.180.82:20128) and off the direct NVIDIA vision endpoint onto 9router, which runs on the same host as both services (127.0.0.1:4014) — loopback avoids the TLS/proxy hop and localhost calls bypass 9router's remote-key guard. - gateway + backend: AI_LLM_BASE_URL default -> http://127.0.0.1:4014/v1 - drop stale 'omniroute' router references from comments/docs now that the active router is 9router (llmClient, llmCaller, ARCHITECTURE, AGENTS) Verified against 9router before wiring: model 'text' -> gemini-3.5-flash-lite (SSE, as the pipeline expects), 'multimodal' -> nemotron-3-nano-omni answers image input, and gemini/gemini-embedding-001 returns 3072 dims — matching the existing Qdrant collections (no reindex needed). The GMW key is already registered in 9router's apiKeys table. typecheck + lint + tests green (gateway 138, backend 37 excluding e2e). --- services/backend/src/shared/config/index.ts | 8 ++++---- services/discord-gateway/AGENTS.md | 2 +- services/discord-gateway/ARCHITECTURE.md | 2 +- .../src/modules/ai-moderation/llmCaller.ts | 2 +- .../src/modules/ai-moderation/llmClient.ts | 4 ++-- services/discord-gateway/src/shared/config/index.ts | 10 ++++++---- 6 files changed, 15 insertions(+), 13 deletions(-) diff --git a/services/backend/src/shared/config/index.ts b/services/backend/src/shared/config/index.ts index a9482fc4..2eaf8636 100644 --- a/services/backend/src/shared/config/index.ts +++ b/services/backend/src/shared/config/index.ts @@ -96,10 +96,10 @@ export const configSchema = z .transform((v) => v === "true") .default(false), AI_LLM_API_KEY: z.string().optional(), - AI_LLM_BASE_URL: z - .string() - .url() - .default("http://100.121.180.82:20128/api/v1"), + // 9router — OpenAI-compatible router on this host (127.0.0.1:4014). + // Loopback on purpose: backend runs on the same machine as 9router, so no + // TLS/proxy hop is needed. + AI_LLM_BASE_URL: z.string().url().default("http://127.0.0.1:4014/v1"), AI_LLM_MODEL: z.string().default("text"), AI_LLM_VISION_MODEL: z.string().optional(), AI_LLM_EMBEDDING_MODEL: z.string().optional(), diff --git a/services/discord-gateway/AGENTS.md b/services/discord-gateway/AGENTS.md index 11ea2233..0c00ea3c 100644 --- a/services/discord-gateway/AGENTS.md +++ b/services/discord-gateway/AGENTS.md @@ -54,7 +54,7 @@ src/ **Never** reintroduce regex/heuristic content classification. 2. **Discord tokens sanitized** before reaching LLM (`discordTokens.ts`). 3. **Semantic cache is batched** — one embed call + one Qdrant batch search. -4. **Streaming is mandatory** against the omniroute base URL. +4. **Streaming is mandatory** against the router base URL. ## AI moderation pipeline diff --git a/services/discord-gateway/ARCHITECTURE.md b/services/discord-gateway/ARCHITECTURE.md index 8043366b..7baf005d 100644 --- a/services/discord-gateway/ARCHITECTURE.md +++ b/services/discord-gateway/ARCHITECTURE.md @@ -194,5 +194,5 @@ pipeline gauges registered by `app/metrics-collector.ts` — numeric snowflake IDs never trigger false positives. - **Semantic cache is batched** (one embed call + one Qdrant batch search), not N sequential round-trips. `ensureQdrantCollection` is memoized. -- **Streaming is mandatory** against the omniroute base URL (non-stream waits for +- **Streaming is mandatory** against the router base URL (non-stream waits for the full body and times out). `llmClient` aggregates SSE chunks. diff --git a/services/discord-gateway/src/modules/ai-moderation/llmCaller.ts b/services/discord-gateway/src/modules/ai-moderation/llmCaller.ts index 373bdef6..7a962cbf 100644 --- a/services/discord-gateway/src/modules/ai-moderation/llmCaller.ts +++ b/services/discord-gateway/src/modules/ai-moderation/llmCaller.ts @@ -92,7 +92,7 @@ export async function callModerationLLM( jsonResponse: { type: "json_object" }, retries: 0, signal, - // Router (omniroute) always streams SSE even when the + // Router always streams SSE even when the // request omits `stream`. In non-stream mode the OpenAI SDK waits // for the FULL body before parsing, so slow/long upstream streams // hit the 30s/60s timeout and abort mid-generation. Streaming mode diff --git a/services/discord-gateway/src/modules/ai-moderation/llmClient.ts b/services/discord-gateway/src/modules/ai-moderation/llmClient.ts index 191008f4..1b853be2 100644 --- a/services/discord-gateway/src/modules/ai-moderation/llmClient.ts +++ b/services/discord-gateway/src/modules/ai-moderation/llmClient.ts @@ -131,7 +131,7 @@ type LLMResponseChunk = { * `delta.content`; falls back to reasoning fields so reasoning-only models * still produce usable aggregated text. Providers differ in the field name: * - DeepSeek-style / Cloudflare gemma → `delta.reasoning_content` - * - mimo (via omniroute) streams reasoning in `delta.reasoning` + + * - mimo (via the router) streams reasoning in `delta.reasoning` + * `delta.reasoning_details[].text` (content:"") — without these fallbacks * vision aggregation came back empty ("Vision API null response"). * Exported for unit tests. @@ -267,7 +267,7 @@ export function buildLlmParams( reasoning: { enabled: false }, // vLLM / Qwen / litellm chat_template_kwargs: { enable_thinking: false }, - // Anthropic / Claude-format (omniroute exposes thinkingFormat + // Anthropic / Claude-format (router exposes thinkingFormat // "claude-adaptive" / "claude-budget" on its reasoning models) thinking: { type: "disabled" }, } as Record); diff --git a/services/discord-gateway/src/shared/config/index.ts b/services/discord-gateway/src/shared/config/index.ts index e779b7a8..1c596935 100644 --- a/services/discord-gateway/src/shared/config/index.ts +++ b/services/discord-gateway/src/shared/config/index.ts @@ -149,10 +149,12 @@ export const configSchema = z .transform((v) => v === "true") .default(false), AI_LLM_API_KEY: z.string().optional(), - AI_LLM_BASE_URL: z - .string() - .url() - .default("http://100.121.180.82:20128/api/v1"), + // 9router — the OpenAI-compatible router on this host (127.0.0.1:4014). + // Loopback on purpose: the gateway runs on the same machine as 9router, + // so no TLS/proxy hop is needed (and localhost bypasses 9router's + // remote-key guard). Public alias https://9router.asepharyana.my.id/v1 + // works too but requires the key for every call. + AI_LLM_BASE_URL: z.string().url().default("http://127.0.0.1:4014/v1"), AI_LLM_MODEL: z.string().default("text"), // Vision uses the SAME router/base URL as text moderation // (AI_LLM_BASE_URL) but a different model alias. The dedicated NVIDIA