Files
GMW/services/discord-gateway/src/shared/config/index.ts
T
asepharyana c7f53e4f7e feat(gateway): remove Jev (System One) analyzer, restore LLM-only text moderation
Jev (oc/jev-1.13-free via 9router /v1/systemone) added as primary text
analyzer was underperforming. Delete the whole feature:
- jevAnalyzer.ts + its unit & live-smoke tests
- Jev-first branch in textBatchProcessor, restore pure callModerationLLM path
- AI_LLM_JEV_* config vars (zod) and .env.example entries
- @typesafe-ai/sdk dependency (+ lockfile)

Behavior: text moderation is LLM-only again, exactly as before the
Jev feature; AGENTS.md invariant 'LLM is the only judge' holds.
2026-09-24 12:06:36 +07:00

450 lines
19 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Unified configuration schema shared by all services.
*
* This is the single source of truth for all environment variables.
* Individual services re-export from here; they do NOT define their own schemas.
*/
import { z } from "zod";
import { ConfigError } from "../errors/index.js";
export const configSchema = z
.object({
// ── Discord ──────────────────────────────────────────────────────────
DISCORD_TOKEN: z
.string()
.min(1, "DISCORD_TOKEN is required")
.transform((value) => value.replace(/^("|')|(?:("|'))$/g, "")),
MONITOR_GUILD_IDS: z
.string()
.default("")
.transform((v) => v.split(",").filter(Boolean)),
MONITOR_GUILD_ID: z.string().min(1).optional(),
TEXT_GUILD_ID: z.string().min(1).optional(),
TEXT_CHANNEL_ID: z.string().min(1).optional(),
EXCLUDED_CHANNEL_IDS: z
.string()
.default("")
.transform((v) => v.split(",").filter(Boolean))
.describe("Channel IDs to exclude from capture"),
EXCLUDED_THREAD_IDS: z
.string()
.default("")
.transform((v) => v.split(",").filter(Boolean))
.describe("Thread IDs to exclude from capture"),
BOT_EXCLUDED_CHANNEL_IDS: z
.string()
.default("1206269771340058694,1318544753821880362")
.transform((v) => v.split(",").filter(Boolean))
.describe(
"Channel IDs where bot messages are NOT captured/analyzed (bot detection stays on everywhere else)",
),
// User IDs whose messages are captured but NEVER AI-analyzed (skip result
// directly, like age-restricted). Used for high-volume music/reaction
// bots that spam the chat log (e.g. Jockie Music) — their now-playing
// embeds carry no moderation signal.
AI_SKIP_ANALYSIS_USER_IDS: z
.string()
.default("411916947773587456")
.transform((v) => v.split(",").filter(Boolean))
.describe("User IDs to skip AI analysis for (captured but not analyzed)"),
AVATAR_SIZE: z.coerce.number().positive().default(64),
// ── Server ───────────────────────────────────────────────────────────
WEBSERVER_PORT: z.coerce.number().positive().default(3001),
NODE_ENV: z
.enum(["development", "production", "test"])
.default("development"),
LOG_LEVEL: z
.enum(["error", "warn", "info", "http", "verbose", "debug", "silly"])
.default("info"),
VERBOSE: z
.string()
.optional()
.transform((v) => v === "true")
.default(false),
ADMIN_PASSWORD: z.string().default("admin123"),
WEBHOOK_URLS: z
.string()
.default("")
.transform((v) => v.split(",").filter(Boolean)),
WEBHOOK_EVENTS: z
.string()
.default("message_flagged,auto_deleted,high_severity")
.transform((v) => v.split(",").filter(Boolean)),
METRICS_PORT: z.coerce.number().positive().default(9090),
// ── Database (PostgreSQL) ────────────────────────────────────────────
DATABASE_URL: z.string().optional(),
POSTGRES_HOST: z.string().default("localhost"),
POSTGRES_PORT: z.coerce.number().int().positive().default(5432),
POSTGRES_USER: z.string().optional(),
POSTGRES_PASSWORD: z.string().optional(),
POSTGRES_DB: z.string().optional(),
// Idle-pool floor. Kept at 0 so the gateway (main + 4 Piscina worker
// threads, each owning its own pg Pool) does not hold ~10 permanently
// open idle connections to PgBouncer. The pool still grows on demand up
// to POSTGRES_POOL_MAX; min:0 only drops idle clients after
// idleTimeoutMillis. This both trims RSS and frees PgBouncer slots.
POSTGRES_POOL_MIN: z.coerce.number().int().min(0).default(0),
// Ceiling for the gateway's per-process pg Pool. Each Piscina worker
// thread owns its own pool (main + 4 workers = 5 pools), so this value
// is the per-thread cap. Kept at 10 (2026-09-09 audit): the real
// bottleneck is PgBouncer's per-(user,db) default_pool_size on imrnes —
// raising this ceiling without raising the Bouncer pool just makes more
// clients queue at the same 20 slots. pool_mode=session means each pg
// Pool client occupies a Bouncer slot for the whole transaction; min:0
// + idleTimeoutMillis frees idle slots automatically.
POSTGRES_POOL_MAX: z.coerce.number().int().positive().default(10),
// ── Redis ────────────────────────────────────────────────────────────
REDIS_URL: z.string().default("redis://localhost:6379"),
// ── Wikipedia (web-search / glossary source) ─────────────────────────
// Native fetch to Wikipedia REST + Action APIs — no SearXNG dependency.
// Language for summaries/search (e.g. "id", "en").
WIKIPEDIA_LANG: z.string().min(1).default("id"),
// Per-request timeout (ms) for Wikipedia API calls.
WIKIPEDIA_TIMEOUT_MS: z.coerce.number().positive().default(8000),
// ── TinyFish web search (fallback when Wikipedia misses) ─────────────
// GET {base}?query=..&location=..&language=.. with X-API-Key header.
// Empty key = fallback disabled (Wikipedia-only, tests stay offline).
TINYFISH_API_KEY: z.string().optional().default(""),
TINYFISH_SEARCH_ENABLED: z
.string()
.optional()
.transform((v) => v === "true")
.default(true),
TINYFISH_SEARCH_BASE_URL: z
.string()
.url()
.default("https://api.search.tinyfish.ai"),
TINYFISH_SEARCH_TIMEOUT_MS: z.coerce.number().positive().default(10000),
TINYFISH_SEARCH_LOCATION: z.string().min(1).default("US"),
TINYFISH_SEARCH_LANGUAGE: z.string().min(1).default("en"),
// ── Connection ───────────────────────────────────────────────────────
RECONNECT_TIMEOUT_MS: z.coerce.number().positive().default(5000),
// ── Attachments ─────────────────────────────────────────────────────
TELE_UPLOAD_URL: z
.string()
.url()
.default("https://upload.asepharyana.my.id/api/upload"),
ATTACHMENT_UPLOAD_TIMEOUT_MS: z.coerce.number().positive().default(30000),
ATTACHMENT_MAX_SIZE_MB: z.coerce.number().positive().default(100),
ATTACHMENT_RETRY_ATTEMPTS: z.coerce.number().positive().default(3),
BACKLOG_SYNC_HOURS: z.coerce.number().positive().default(24),
BACKLOG_SYNC_BATCH_SIZE: z.coerce
.number()
.int()
.positive()
.max(100)
.default(100),
// ── AI Analysis ─────────────────────────────────────────────────────
AI_ANALYSIS_ENABLED: z
.string()
.optional()
.transform((v) => v === "true")
.default(false),
AI_LLM_API_KEY: z.string().optional(),
AI_LLM_BASE_URL: z
.string()
.url()
.default("http://100.121.180.82:20128/api/v1"),
AI_LLM_MODEL: z.string().default("text"),
// Vision uses the SAME router/base URL as text moderation
// (AI_LLM_BASE_URL) but a different model alias. The dedicated NVIDIA
// multimodal endpoint was removed.
AI_LLM_VISION_MODEL: z.string().default("multimodal"),
AI_LLM_DISABLE_THINKING: z
.string()
.default("true")
.transform((v) => v === "true")
.describe(
"Disable LLM chain-of-thought (reasoning/thinking) to speed up AI analysis. Set false to restore thinking.",
),
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
AI_LLM_EMBEDDING_MIN_SIMILARITY: z.coerce
.number()
.min(0)
.max(1)
.default(0.97),
// Two-band semantic acceptance (2026-08-24): non-actionable verdicts
// (clean, no flags, action=none) may be reused from a LOOSER similarity
// band than actionable ones (warn/flagged). Actionable verdicts keep the
// strict gate above; anything between the two bands falls through to the
// LLM (fail-open toward accuracy).
AI_LLM_EMBEDDING_MIN_SIMILARITY_CLEAN: z.coerce
.number()
.min(0)
.max(1)
.default(0.92),
AI_LLM_EMBEDDING_MAX_CANDIDATES: z.coerce
.number()
.int()
.positive()
.default(50),
// Qdrant vector store for the semantic moderation cache. When
// QDRANT_URL is set, embeddings are stored/searched there (Postgres
// embedding column remains as a legacy fallback).
QDRANT_URL: z.string().optional(),
QDRANT_COLLECTION: z.string().default("gmw_text_moderation"),
QDRANT_ARCHIVE_COLLECTION: z.string().default("gmw_message_archive"),
QDRANT_API_KEY: z.string().optional(),
AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(8),
AI_LLM_IMAGE_MAX_DIMENSION: z.coerce
.number()
.int()
.positive()
.default(1024),
AI_LLM_TEXT_BATCH_SIZE: z.coerce.number().int().positive().default(60),
AI_LLM_MAX_COMPLETION_TOKENS: z.coerce
.number()
.int()
.positive()
.default(16384),
AI_LLM_MEDIA_ANALYSIS_TIMEOUT_MS: z.coerce
.number()
.int()
.positive()
.default(120_000),
// Standalone image/sticker/emoji vision analysis (analyzeSingleMediaImage
// → llmVision → llmChat). Decoupled from the media *batch* timeout above so
// a single vision call can be tuned independently. 2 minutes by default —
// vision models (especially behind a router) need headroom for large images
// and the media batch budget grew to 120s (2026-09-09) so single-image calls
// must not be the bottleneck in the fallback chain.
AI_LLM_VISION_ANALYSIS_TIMEOUT_MS: z.coerce
.number()
.int()
.positive()
.default(120_000),
// Text-only moderation batches are cheaper than media (no downloads /
// vision pre-pass), so they get their own (shorter) timeout instead of
// being tied to the media budget. Raised 45s→75s (2026-09-09): the text
// model behind the router regularly exceeds 45s on long context batches,
// and the individual-fallback re-run adds another full timeout cycle
// before marking the message exhausted. 75s is still bounded and keeps
// the status queue from piling up.
AI_LLM_TEXT_ANALYSIS_TIMEOUT_MS: z.coerce
.number()
.int()
.positive()
.default(75_000),
// Term glossary — per-word Wikipedia lookups (via SearXNG) for words the
// LLM may not know (slang, jargon, regional language, foreign terms).
// Definitions are cached (in-memory + Redis) so repeat lookups are fast.
// Disable to skip glossary lookups entirely and analyze without them.
AI_GLOSSARY_ENABLED: z
.string()
.optional()
.transform((v) => v === "true")
.default(true),
// Max glossary terms looked up per analysis batch (keeps latency bounded).
AI_GLOSSARY_MAX_TERMS: z.coerce.number().int().min(1).max(20).default(6),
// Per-user personal profile summaries (userProfileLearner). Disabled by
// default: profiles bloat the analysis context and add LLM/DB cost for
// little moderation signal — user history context (last flagged messages)
// is injected via <user_history> instead of a numeric trust score.
AI_USER_PROFILE_LEARNING_ENABLED: z
.string()
.optional()
.transform((v) => v === "true")
.default(false),
// Min word length for a term to be considered glossary-worthy.
AI_GLOSSARY_MIN_WORD_LENGTH: z.coerce
.number()
.int()
.min(2)
.max(20)
.default(5),
// ── AI Analysis Timing ──────────────────────────────────────────────
AI_ANALYSIS_DEBOUNCE_MS: z.coerce.number().positive().default(250),
AI_ANALYSIS_RECOVERY_INTERVAL_MS: z.coerce
.number()
.positive()
.default(10000),
AI_ANALYSIS_ERROR_COOLDOWN_MS: z.coerce.number().positive().default(30000),
// Upload-pending batch poll (2026-08-25): when a batch is deferred because
// attachments are still uploading, the processor re-schedules with this
// base delay (linear ramp per consecutive poll, capped) instead of the
// 250ms debounce — the old path hot-looped ~300ms for the whole upload.
AI_ANALYSIS_UPLOAD_POLL_MS: z.coerce.number().positive().default(1500),
AI_ANALYSIS_MAX_UPLOAD_POLL_MS: z.coerce.number().positive().default(8000),
// ── AI Analysis Batch ───────────────────────────────────────────────
AI_ANALYSIS_MAX_BATCH_SIZE: z.coerce.number().int().positive().default(200),
// Global exact-cache reuse guard (2026-08-24): a context-scoped miss may
// fall back to the legacy bare (context-free) key, but ONLY for verdicts
// that cannot trigger an action and are fresh + confident. These knobs
// bound that reuse.
AI_CACHE_GLOBAL_REUSE_MIN_CONFIDENCE: z.coerce
.number()
.min(0)
.max(1)
.default(0.85),
AI_CACHE_GLOBAL_REUSE_MAX_AGE_H: z.coerce
.number()
.int()
.positive()
.default(120),
AI_ANALYSIS_MAX_CONTEXT_TOKENS: z.coerce.number().positive().default(8000),
AI_ANALYSIS_MAX_TARGET_TOKENS: z.coerce.number().positive().default(14000),
AI_ANALYSIS_CONTEXT_MESSAGE_LIMIT: z.coerce
.number()
.int()
.positive()
.default(20),
// Recency gates for conversation context. A silence longer than GAP_MS
// between context messages = the conversation restarted (older messages
// dropped); MAX_AGE_MS caps how far back context is considered relevant.
AI_ANALYSIS_CONTEXT_GAP_MS: z.coerce
.number()
.positive()
.default(12 * 60 * 1000),
AI_ANALYSIS_CONTEXT_MAX_AGE_MS: z.coerce
.number()
.positive()
.default(45 * 60 * 1000),
AI_ANALYSIS_PROCESSING_TIMEOUT_MS: z.coerce
.number()
.positive()
.default(120000),
AI_ANALYSIS_INDIVIDUAL_MAX_CONCURRENT: z.coerce
.number()
.int()
.positive()
.default(50),
AI_ANALYSIS_INDIVIDUAL_CB_THRESHOLD: z.coerce
.number()
.int()
.positive()
.default(50),
// Text-analysis worker pool size (2026-08-31: split from the media pool
// below so a slow image/vision batch can never occupy every thread and
// starve the far more common text-only batches). Default 4 (not
// availableParallelism) because each Piscina thread owns its own
// pLimit(5) semaphore — on big VPSes availableParallelism × 5 concurrent
// LLM calls would overwhelm the router. Keep threads modest; concurrency
// is capped per-thread anyway.
PISCINA_MAX_THREADS: z.coerce.number().int().positive().default(4),
// Media-analysis worker pool size — dedicated threads for batches that
// contain images/stickers/embeds (download + vision + LLM, much slower
// than text). Kept small since media batches are less frequent and each
// one is long-running; sized independently from PISCINA_MAX_THREADS so
// tuning one never starves the other.
PISCINA_MEDIA_MAX_THREADS: z.coerce.number().int().positive().default(2),
// ── Auto Delete ─────────────────────────────────────────────────────
AUTO_DELETE_FLAGGED_ENABLED: z
.string()
.optional()
.transform((v) => v === "true")
.default(true),
AUTO_DELETE_FLAGGED_DRY_RUN: z
.string()
.optional()
.transform((v) => v === "true")
.default(false),
AUTO_DELETE_FLAGGED_DELAY_MS: z.coerce.number().min(0).default(0),
AUTO_DELETE_MIN_CONFIDENCE: z.coerce.number().min(0).max(1).default(0.5),
AUTO_DELETE_ALLOWED_SEVERITIES: z
.string()
.default("critical,high,medium,low"),
AUTO_DELETE_ALLOWED_CATEGORIES: z.string().default(""),
AUTO_DELETE_EXCLUDED_CHANNEL_IDS: z.string().default(""),
AUTO_DELETE_EXCLUDED_USER_IDS: z.string().default(""),
AUTO_DELETE_NOTIFY_USER: z
.string()
.optional()
.transform((v) => v === "true")
.default(false),
AUTO_DELETE_LOG_CHANNEL_ID: z.string().default(""),
// ── Nickname Reset (offensive_username enforcement) ────────────────
// When the only violation is the member's server nickname, reset the
// nickname to the default username instead of deleting the message.
AUTO_NICKNAME_RESET_ENABLED: z
.string()
.optional()
.transform((v) => v === "true")
.default(true),
AUTO_NICKNAME_RESET_COOLDOWN_MS: z.coerce
.number()
.positive()
.default(10 * 60 * 1000),
// ── Retention ───────────────────────────────────────────────────────
RETENTION_MESSAGES_DAYS: z.coerce.number().int().min(0).default(0),
RETENTION_ATTACHMENTS_DAYS: z.coerce.number().int().min(0).default(0),
RETENTION_CLEANUP_INTERVAL_MS: z.coerce
.number()
.positive()
.default(24 * 60 * 60 * 1000),
RETENTION_DRY_RUN: z
.string()
.optional()
.transform((v) => v === "true")
.default(true),
AUTO_MIGRATE_ON_STARTUP: z
.string()
.optional()
.transform((v) => v === "true")
.default(true),
})
.superRefine((value, ctx) => {
if (!value.AI_ANALYSIS_ENABLED) {
// skip: AI analysis not enabled
} else if (!value.AI_LLM_API_KEY) {
ctx.addIssue({
code: z.ZodIssueCode.custom,
path: ["AI_LLM_API_KEY"],
message: "AI_LLM_API_KEY is required when AI_ANALYSIS_ENABLED=true",
});
}
// Validate database configuration
if (!value.DATABASE_URL && !value.POSTGRES_HOST) {
ctx.addIssue({
code: z.ZodIssueCode.custom,
path: ["DATABASE_URL"],
message: "Either DATABASE_URL or POSTGRES_HOST must be provided",
});
}
});
export type AppConfig = z.infer<typeof configSchema> & {
EFFECTIVE_TEXT_GUILD_ID?: string;
EFFECTIVE_MONITOR_GUILD_IDS: string[];
};
export function loadConfig(env: NodeJS.ProcessEnv = process.env): AppConfig {
try {
const parsed = configSchema.parse(env);
return {
...parsed,
EFFECTIVE_TEXT_GUILD_ID: parsed.TEXT_GUILD_ID ?? parsed.MONITOR_GUILD_ID,
EFFECTIVE_MONITOR_GUILD_IDS:
parsed.MONITOR_GUILD_IDS.length > 0
? parsed.MONITOR_GUILD_IDS
: parsed.MONITOR_GUILD_ID
? [parsed.MONITOR_GUILD_ID]
: [],
};
} catch (error) {
if (error instanceof z.ZodError) {
const messages = error.issues
.map((e) => `${e.path.join(".")}: ${e.message}`)
.join("\n");
throw new ConfigError(`Configuration validation failed:\n${messages}`);
}
throw error;
}
}
/** Singleton config loaded from process.env at import time. */
export const config = loadConfig();