chore: migrate AI LLM router from 9router to omniroute
Switch GMW's AI LLM base URL from 9router (https://9router.asepharyana.my.id/v1) to omniroute on imrnes (http://100.121.180.82:20128/api/v1). - Update default AI_LLM_BASE_URL in discord-gateway + backend config schemas - Update .env.example documentation - Update all 9router references in comments/docs/tests to omniroute - Production BWS secret gmw_ai_llm_base_url already updated Omniroute uses /api/v1 prefix (not /v1 like 9router), so the base URL now correctly points at the right API path for the OpenAI SDK.
This commit is contained in:
+1
-1
@@ -85,7 +85,7 @@ BACKLOG_SYNC_BATCH_SIZE=100 # Messages per backlog batch, max 100 (d
|
|||||||
# === AI Analysis ===
|
# === AI Analysis ===
|
||||||
AI_ANALYSIS_ENABLED=false # Enable AI content moderation (default: false)
|
AI_ANALYSIS_ENABLED=false # Enable AI content moderation (default: false)
|
||||||
AI_LLM_API_KEY= # REQUIRED if AI_ANALYSIS_ENABLED=true. LLM API key
|
AI_LLM_API_KEY= # REQUIRED if AI_ANALYSIS_ENABLED=true. LLM API key
|
||||||
AI_LLM_BASE_URL=https://9router.asepharyana.my.id/v1 # LLM API base URL (9router — OpenAI-compatible router, replaces omniroute)
|
AI_LLM_BASE_URL=http://100.121.180.82:20128/api/v1 # LLM API base URL (omniroute — OpenAI-compatible router on imrnes, Tailscale 100.121.180.82)
|
||||||
AI_LLM_MODEL=text # LLM text model name (default: text)
|
AI_LLM_MODEL=text # LLM text model name (default: text)
|
||||||
# AI_LLM_VISION_MODEL= # Vision model for image analysis (falls back to AI_LLM_MODEL)
|
# AI_LLM_VISION_MODEL= # Vision model for image analysis (falls back to AI_LLM_MODEL)
|
||||||
# AI_LLM_EMBEDDING_MODEL= # Embedding model for semantic moderation cache (optional; enables near-duplicate text reuse to save LLM calls)
|
# AI_LLM_EMBEDDING_MODEL= # Embedding model for semantic moderation cache (optional; enables near-duplicate text reuse to save LLM calls)
|
||||||
|
|||||||
@@ -158,7 +158,7 @@ Gaya ngobrol:
|
|||||||
tool_choice: "auto",
|
tool_choice: "auto",
|
||||||
max_tokens: 600,
|
max_tokens: 600,
|
||||||
temperature: 0.4,
|
temperature: 0.4,
|
||||||
// Non-streaming: request a single complete response. 9router may
|
// Non-streaming: request a single complete response. omniroute may
|
||||||
// still emit SSE even with stream:false, so the parser below
|
// still emit SSE even with stream:false, so the parser below
|
||||||
// handle both raw-JSON and SSE bodies.
|
// handle both raw-JSON and SSE bodies.
|
||||||
stream: false,
|
stream: false,
|
||||||
@@ -176,7 +176,7 @@ Gaya ngobrol:
|
|||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
|
||||||
// Parse the body into content + tool_calls. 9router may return either
|
// Parse the body into content + tool_calls. omniroute may return either
|
||||||
// a single JSON object (stream:false honored) or SSE text (stream
|
// a single JSON object (stream:false honored) or SSE text (stream
|
||||||
// implied) — parseResponse handles both.
|
// implied) — parseResponse handles both.
|
||||||
const { content, toolCalls } = this.parseResponse(
|
const { content, toolCalls } = this.parseResponse(
|
||||||
@@ -252,7 +252,7 @@ Gaya ngobrol:
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Parse an LLM HTTP body into content + tool_calls. Handles both shapes
|
* Parse an LLM HTTP body into content + tool_calls. Handles both shapes
|
||||||
* 9router can return: a single JSON object (stream:false honored) or SSE
|
* omniroute can return: a single JSON object (stream:false honored) or SSE
|
||||||
* text (stream implied). For SSE we delegate to parseSse.
|
* text (stream implied). For SSE we delegate to parseSse.
|
||||||
*/
|
*/
|
||||||
private parseResponse(body: string): {
|
private parseResponse(body: string): {
|
||||||
@@ -306,7 +306,7 @@ Gaya ngobrol:
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Parse an SSE stream body into accumulated content + any tool_calls.
|
* Parse an SSE stream body into accumulated content + any tool_calls.
|
||||||
* 9router (and most OpenAI-compatible routers) emit `data: {json}` lines
|
* omniroute (and most OpenAI-compatible routers) emit `data: {json}` lines
|
||||||
* even when stream is only implied; we must collect deltas manually.
|
* even when stream is only implied; we must collect deltas manually.
|
||||||
*/
|
*/
|
||||||
private parseSse(body: string): {
|
private parseSse(body: string): {
|
||||||
|
|||||||
@@ -131,7 +131,7 @@ export const configSchema = z
|
|||||||
AI_LLM_BASE_URL: z
|
AI_LLM_BASE_URL: z
|
||||||
.string()
|
.string()
|
||||||
.url()
|
.url()
|
||||||
.default("https://9router.asepharyana.my.id/v1"),
|
.default("http://100.121.180.82:20128/api/v1"),
|
||||||
AI_LLM_MODEL: z.string().default("text"),
|
AI_LLM_MODEL: z.string().default("text"),
|
||||||
AI_LLM_VISION_MODEL: z.string().optional(),
|
AI_LLM_VISION_MODEL: z.string().optional(),
|
||||||
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
|
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
|
||||||
|
|||||||
@@ -139,5 +139,5 @@ pipeline gauges — `ai_analysis_queued_conversations`,
|
|||||||
numeric snowflake IDs never trigger false positives.
|
numeric snowflake IDs never trigger false positives.
|
||||||
- **Semantic cache is batched** (one embed call + one Qdrant batch search),
|
- **Semantic cache is batched** (one embed call + one Qdrant batch search),
|
||||||
not N sequential round-trips. `ensureQdrantCollection` is memoized.
|
not N sequential round-trips. `ensureQdrantCollection` is memoized.
|
||||||
- **Streaming is mandatory** against the 9router base URL (non-stream waits for
|
- **Streaming is mandatory** against the omniroute base URL (non-stream waits for
|
||||||
the full body and times out). `llmClient` aggregates SSE chunks.
|
the full body and times out). `llmClient` aggregates SSE chunks.
|
||||||
|
|||||||
@@ -85,7 +85,7 @@ export async function callModerationLLM(
|
|||||||
jsonResponse: { type: "json_object" },
|
jsonResponse: { type: "json_object" },
|
||||||
retries: 0,
|
retries: 0,
|
||||||
signal,
|
signal,
|
||||||
// Router (9router / formerly omniroute) always streams SSE even when the
|
// Router (omniroute) always streams SSE even when the
|
||||||
// request omits `stream`. In non-stream mode the OpenAI SDK waits
|
// request omits `stream`. In non-stream mode the OpenAI SDK waits
|
||||||
// for the FULL body before parsing, so slow/long upstream streams
|
// for the FULL body before parsing, so slow/long upstream streams
|
||||||
// hit the 30s/60s timeout and abort mid-generation. Streaming mode
|
// hit the 30s/60s timeout and abort mid-generation. Streaming mode
|
||||||
|
|||||||
@@ -94,7 +94,7 @@ type LLMResponseChunk = {
|
|||||||
* `delta.content`; falls back to reasoning fields so reasoning-only models
|
* `delta.content`; falls back to reasoning fields so reasoning-only models
|
||||||
* still produce usable aggregated text. Providers differ in the field name:
|
* still produce usable aggregated text. Providers differ in the field name:
|
||||||
* - DeepSeek-style / Cloudflare gemma → `delta.reasoning_content`
|
* - DeepSeek-style / Cloudflare gemma → `delta.reasoning_content`
|
||||||
* - mimo (via 9router) streams reasoning in `delta.reasoning` +
|
* - mimo (via omniroute) streams reasoning in `delta.reasoning` +
|
||||||
* `delta.reasoning_details[].text` (content:"") — without these fallbacks
|
* `delta.reasoning_details[].text` (content:"") — without these fallbacks
|
||||||
* vision aggregation came back empty ("Vision API null response").
|
* vision aggregation came back empty ("Vision API null response").
|
||||||
* Exported for unit tests.
|
* Exported for unit tests.
|
||||||
@@ -225,7 +225,7 @@ export function buildLlmParams(
|
|||||||
reasoning: { enabled: false },
|
reasoning: { enabled: false },
|
||||||
// vLLM / Qwen / litellm
|
// vLLM / Qwen / litellm
|
||||||
chat_template_kwargs: { enable_thinking: false },
|
chat_template_kwargs: { enable_thinking: false },
|
||||||
// Anthropic / Claude-format (9router exposes thinkingFormat
|
// Anthropic / Claude-format (omniroute exposes thinkingFormat
|
||||||
// "claude-adaptive" / "claude-budget" on its reasoning models)
|
// "claude-adaptive" / "claude-budget" on its reasoning models)
|
||||||
thinking: { type: "disabled" },
|
thinking: { type: "disabled" },
|
||||||
} as Record<string, unknown>);
|
} as Record<string, unknown>);
|
||||||
|
|||||||
@@ -149,7 +149,7 @@ export const configSchema = z
|
|||||||
AI_LLM_BASE_URL: z
|
AI_LLM_BASE_URL: z
|
||||||
.string()
|
.string()
|
||||||
.url()
|
.url()
|
||||||
.default("https://9router.asepharyana.my.id/v1"),
|
.default("http://100.121.180.82:20128/api/v1"),
|
||||||
AI_LLM_MODEL: z.string().default("text"),
|
AI_LLM_MODEL: z.string().default("text"),
|
||||||
// Vision uses the SAME router/base URL as text moderation
|
// Vision uses the SAME router/base URL as text moderation
|
||||||
// (AI_LLM_BASE_URL) but a different model alias. The dedicated NVIDIA
|
// (AI_LLM_BASE_URL) but a different model alias. The dedicated NVIDIA
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
// ═══════════════════════════════════════════════════════════════════════════
|
// ═══════════════════════════════════════════════════════════════════════════
|
||||||
// llmClient chunk extraction — reasoning_content fallback (pure, no network)
|
// llmClient chunk extraction — reasoning_content fallback (pure, no network)
|
||||||
// ═══════════════════════════════════════════════════════════════════════════
|
// ═══════════════════════════════════════════════════════════════════════════
|
||||||
// Regression: 9router "multimodal" combo routed to cloudflare gemma-4-26b
|
// Regression: omniroute "multimodal" combo routed to cloudflare gemma-4-26b
|
||||||
// which streams ALL output in delta.reasoning_content with content:"" — the
|
// which streams ALL output in delta.reasoning_content with content:"" — the
|
||||||
// old extractor returned empty text → llmVision reported "Vision API null
|
// old extractor returned empty text → llmVision reported "Vision API null
|
||||||
// response" → every image moderation batch fell back to text-only analysis
|
// response" → every image moderation batch fell back to text-only analysis
|
||||||
@@ -19,7 +19,7 @@ describe("extractChunkText — streaming chunk text extraction", () => {
|
|||||||
});
|
});
|
||||||
|
|
||||||
it("falls back to delta.reasoning_content when content is empty — reasoning-only models (cloudflare gemma)", () => {
|
it("falls back to delta.reasoning_content when content is empty — reasoning-only models (cloudflare gemma)", () => {
|
||||||
// Exact shape seen from 9router → cloudflare-ai/@cf/google/gemma-4-26b:
|
// Exact shape seen from omniroute → cloudflare-ai/@cf/google/gemma-4-26b:
|
||||||
// {"choices":[{"delta":{"content":"","reasoning_content":"Task","role":"assistant"},"finish_reason":null,...}]}
|
// {"choices":[{"delta":{"content":"","reasoning_content":"Task","role":"assistant"},"finish_reason":null,...}]}
|
||||||
expect(
|
expect(
|
||||||
extractChunkText({
|
extractChunkText({
|
||||||
@@ -33,8 +33,8 @@ describe("extractChunkText — streaming chunk text extraction", () => {
|
|||||||
).toBe("Task");
|
).toBe("Task");
|
||||||
});
|
});
|
||||||
|
|
||||||
it('falls back to delta.reasoning — mimo via 9router streams reasoning there with content:""', () => {
|
it('falls back to delta.reasoning — mimo via omniroute streams reasoning there with content:""', () => {
|
||||||
// Exact shape seen from 9router → mimo-v2.5-free (2026-08-11):
|
// Exact shape seen from omniroute → mimo-v2.5-free (2026-08-11):
|
||||||
// {"choices":[{"delta":{"content":"","reasoning":"The user wants a","role":"assistant"},"finish_reason":null,...}]}
|
// {"choices":[{"delta":{"content":"","reasoning":"The user wants a","role":"assistant"},"finish_reason":null,...}]}
|
||||||
expect(
|
expect(
|
||||||
extractChunkText({
|
extractChunkText({
|
||||||
|
|||||||
@@ -6,7 +6,7 @@
|
|||||||
// a VALID vision_llm result. Every later analysis of the same image (same
|
// a VALID vision_llm result. Every later analysis of the same image (same
|
||||||
// hash / phash) then hit the poisoned cache and the moderation LLM wrote
|
// hash / phash) then hit the poisoned cache and the moderation LLM wrote
|
||||||
// "lampiran yang gagal terbaca" — image analysis seemed permanently broken
|
// "lampiran yang gagal terbaca" — image analysis seemed permanently broken
|
||||||
// even though 9router was responding fine.
|
// even though omniroute was responding fine.
|
||||||
import { describe, expect, it } from "vitest";
|
import { describe, expect, it } from "vitest";
|
||||||
import { isNoImageSeenText } from "../src/modules/ai-moderation/visionAnalyzer.js";
|
import { isNoImageSeenText } from "../src/modules/ai-moderation/visionAnalyzer.js";
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user