From 43e35a71dd004ea547cf46fc90d0cae388a319d4 Mon Sep 17 00:00:00 2001 From: asepharyana Date: Fri, 18 Sep 2026 21:27:06 +0700 Subject: [PATCH] test: add live-LLM E2E moderation test suite MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add tests/llmE2e.test.ts — 7 end-to-end tests driving the REAL moderation prompt pipeline (buildSystemPrompt → XML payload → llmChat → parseModerationResponse) against a live model via omniroute. Covers: clean technical content (no false positives), harassment (flagged), username-only offenses including 'Pecinta Pria' + sexual/provocative usernames + SARA-in-username (always warn/low, NEVER delete — the nickname-reset path), and spam bursts. Gated behind AI_LLM_BASE_URL + AI_LLM_API_KEY: CI (no creds) skips the file → 216 unit tests stay green, zero LLM cost. Run locally via pnpm test:e2e:live (scripts/run-llm-e2e.sh injects creds from bws). Verified: 223/223 tests pass with live LLM, stability across 4 runs, typecheck + biome clean. docs: TESTING.md. ignore .hermes/ plans. --- .gitignore | 1 + services/discord-gateway/TESTING.md | 86 +++++++ services/discord-gateway/package.json | 5 +- .../discord-gateway/scripts/run-llm-e2e.sh | 35 +++ services/discord-gateway/tests/llmE2e.test.ts | 215 ++++++++++++++++++ 5 files changed, 341 insertions(+), 1 deletion(-) create mode 100644 services/discord-gateway/TESTING.md create mode 100755 services/discord-gateway/scripts/run-llm-e2e.sh create mode 100644 services/discord-gateway/tests/llmE2e.test.ts diff --git a/.gitignore b/.gitignore index f9a08e7e..1e6c8e0c 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,7 @@ services/frontend/frontend/dist/ public/app/ .muxer-queue.** .claude/ +.hermes/ .env.test logs/ .codegraph/ diff --git a/services/discord-gateway/TESTING.md b/services/discord-gateway/TESTING.md new file mode 100644 index 00000000..a9bcc3aa --- /dev/null +++ b/services/discord-gateway/TESTING.md @@ -0,0 +1,86 @@ +# Testing — discord-gateway + +Two test tiers, both in `tests/` and both run by `vitest`: + +| Tier | Files | What it proves | Runs in CI | Cost | +| --- | --- | --- | --- | --- | +| **Unit** | 25 files (216 tests) | Pure logic: prompt builders, parsers, cache guards, eligibility routing, dedup, media keying | ✅ always | free | +| **E2E (live LLM)** | `tests/llmE2e.test.ts` (7 tests) | The **real** moderation prompt + **real** model produce correct verdicts end-to-end | ⏭️ skipped (no creds) | ~30s, 7 LLM calls | + +## Commands + +```bash +pnpm test # everything; E2E auto-skips when creds absent +pnpm test:unit # unit only (fast, no network) +pnpm test:e2e # E2E only (skips if creds absent) +pnpm test:e2e:live # E2E with live creds injected from Bitwarden (host only) +``` + +## What the E2E tier covers + +`tests/llmE2e.test.ts` drives the exact production path: + +``` +buildSystemPrompt({ mode: "text" }) ← real system rules + output schema + ↓ + XML payload ← same shape textBatchProcessor sends + ↓ +llmChat(...) ← real model via omniroute + ↓ +parseModerationResponse(raw, ids) ← real Zod schema + severity/action derivation + ↓ +assertions on status / flags / severity / recommendedAction +``` + +Cases: + +1. **Clean technical question** → `clean`, no `threat` flag, no delete. +2. **Physics/engineering discussion** → `clean`; guards against false-positive `threat`/`violence`. +3. **Explicit harassment + death threat** → flagged, non-empty flags. +4. **`Pecinta Pria` username + clean content** → **never delete**, never high/critical — the nickname-reset path. +5. **Sexual/provocative usernames + clean content** → never delete, never high/critical. +6. **SARA term in username only + clean content** → never delete — username is identity, not a forbidden-topic discussion. +7. **Repeated short message (`repetitions="5"`)** → spam handling stays in the warn/flag band. + +### Why assertions are bands, not exact matches + +Real models are non-deterministic. Pinning exact JSON would make the suite flaky and would +test the model, not the prompt. Each assertion instead encodes an **invariant the prompt +guarantees** — "username-only offense never deletes", "clean technical text is never a threat". +A regression in `prompts/rules.ts` or `prompts/output.ts` that breaks one of those invariants +fails the E2E tier. + +### Flakiness handling + +The `moderate()` helper retries a malformed response once, mirroring production: `llmClient` +has `DEFAULT_RETRIES = 2` and `aiAnalyzer`'s recovery worker re-analyses messages left in +`error`/`analysis_incomplete`. Observed otherwise: an occasional degenerate stream +(`results` as strings) fails Zod. Production recovers; the test retries the same way. + +## Gating (why CI stays green and free) + +```ts +const HAS_LLM = Boolean(process.env.AI_LLM_BASE_URL && process.env.AI_LLM_API_KEY); +const runIfLLM = HAS_LLM ? describe : describe.skip; +``` + +CI runs `vitest run` with no LLM env → the file reports `1 skipped`, 7 tests skipped, +zero network calls. Run locally with creds for the full signal. + +## Running E2E with live credentials + +```bash +pnpm test:e2e:live # reads /etc/bws-token → bws-env gmw → AI_LLM_* vars +``` + +Or manually: + +```bash +export AI_LLM_BASE_URL=http:///api/v1 +export AI_LLM_API_KEY= +pnpm test:e2e +``` + +Do **not** add LLM credentials to CI secrets: the E2E tier calls a paid model and asserts on +non-deterministic output, so a red run would be ambiguous. It is a deliberate local/pre-release +gate; CI owns the deterministic unit tier. diff --git a/services/discord-gateway/package.json b/services/discord-gateway/package.json index 9c882ff5..a9304f18 100644 --- a/services/discord-gateway/package.json +++ b/services/discord-gateway/package.json @@ -18,7 +18,10 @@ "typecheck": "tsc --noEmit", "lint": "biome check --diagnostic-level=error .", "format": "biome format --write .", - "test": "vitest run" + "test": "vitest run", + "test:unit": "vitest run --exclude \"tests/llmE2e.test.ts\"", + "test:e2e": "vitest run tests/llmE2e.test.ts", + "test:e2e:live": "bash scripts/run-llm-e2e.sh" }, "dependencies": { "@discordjs/opus": "^0.10.0", diff --git a/services/discord-gateway/scripts/run-llm-e2e.sh b/services/discord-gateway/scripts/run-llm-e2e.sh new file mode 100755 index 00000000..d91ca369 --- /dev/null +++ b/services/discord-gateway/scripts/run-llm-e2e.sh @@ -0,0 +1,35 @@ +#!/bin/bash +# ──────────────────────────────────────────────────────────────────────────── +# Run the LLM E2E moderation tests against the REAL model. +# +# These tests are gated behind AI_LLM_BASE_URL + AI_LLM_API_KEY, so plain +# `pnpm test` / CI skips them (zero cost, stays green). This script injects +# the gateway's live credentials from Bitwarden Secrets Manager and runs them. +# +# Usage (on the host that runs the gateway): +# bash scripts/run-llm-e2e.sh # all E2E tests, verbose +# bash scripts/run-llm-e2e.sh --watch # extra args go to vitest +# +# Requires: sudo access to `bws-env gmw` + /etc/bws-token. +# Cost: 7 real LLM calls (~30s, a few thousand tokens) per run. +# ──────────────────────────────────────────────────────────────────────────── +set -euo pipefail + +cd "$(dirname "$0")/.." + +if [ ! -r /etc/bws-token ]; then + echo "ERROR: /etc/bws-token not readable — cannot fetch live LLM credentials." >&2 + echo " Set AI_LLM_BASE_URL + AI_LLM_API_KEY manually instead." >&2 + exit 1 +fi + +export BWS_ACCESS_TOKEN="$(tr -d '\r\n' < /etc/bws-token)" + +sudo -n env BWS_ACCESS_TOKEN="$BWS_ACCESS_TOKEN" bash -c ' + set -euo pipefail + export BWS_ACCESS_TOKEN="$BWS_ACCESS_TOKEN" + ENV=$(bws-env gmw 2>/dev/null) + set -a; eval "$ENV"; set +a + cd '"$PWD"' + exec ./node_modules/.bin/vitest run tests/llmE2e.test.ts --reporter=verbose "$@" +' -- "$@" diff --git a/services/discord-gateway/tests/llmE2e.test.ts b/services/discord-gateway/tests/llmE2e.test.ts new file mode 100644 index 00000000..1a90d0af --- /dev/null +++ b/services/discord-gateway/tests/llmE2e.test.ts @@ -0,0 +1,215 @@ +// ═══════════════════════════════════════════════════════════════════════════ +// LLM E2E test — real model, full prompt pipeline +// +// Validates the REAL moderation prompt (system rules + output schema + +// few-shots) end-to-end against a live LLM: +// +// buildSystemPrompt() → XML → llmChat → +// parseModerationResponse() → assert on verdicts. +// +// Gate: skipped unless AI_LLM_BASE_URL + AI_LLM_API_KEY are set. CI runs +// without them → these tests no-op there (no token cost). Run locally with +// the gateway's live env: +// +// BWS_ACCESS_TOKEN=$(tr -d '\r\n' < /etc/bws-token) +// ENV=$(sudo bws-env gmw); set -a; eval "$ENV"; set +a +// npx vitest run tests/llmE2e.test.ts +// +// Assertions are DELIBERATELY relaxed (flag presence, severity direction, +// recommended action category) — real LLMs are non-deterministic. This test +// catches REGRESSIONS in prompt rules (e.g. a username-only offense suddenly +// producing `delete`, or a clean technical message flagging as threat), not +// exact-string matching. +// ═══════════════════════════════════════════════════════════════════════════ +import { describe, expect, it } from "vitest"; +import { llmChat } from "../src/modules/ai-moderation/llmClient.js"; +import { parseModerationResponse } from "../src/modules/ai-moderation/moderationResponseParser.js"; +import { buildSystemPrompt } from "../src/modules/ai-moderation/prompts/system.js"; + +// ── Gate: only run when a real LLM is configured ────────────────────────── +const HAS_LLM = Boolean( + process.env.AI_LLM_BASE_URL && process.env.AI_LLM_API_KEY, +); +const runIfLLM = HAS_LLM ? describe : describe.skip; + +/** Build the user payload exactly like textBatchProcessor does. */ +function buildUserPayload( + messages: Array<{ + id: string; + user: string; + content: string; + repetitions?: string; + }>, +): string { + const block = messages + .map((m) => { + const repAttr = m.repetitions ? ` repetitions="${m.repetitions}"` : ""; + return `\n ${m.content}\n`; + }) + .join("\n"); + return `\n${block}\n`; +} + +async function moderate( + messages: Array<{ + id: string; + user: string; + content: string; + repetitions?: string; + }>, + opts: { maxTokens?: number } = {}, +): Promise> { + const system = buildSystemPrompt({ mode: "text" }); + const user = buildUserPayload(messages); + // Like production (aiAnalyzer recovery loop + llmCaller retry), tolerate + // one malformed response and re-ask. Real LLM streams occasionally return + // degenerate JSON; prod retries those messages in the recovery worker. + let lastError: unknown; + for (let attempt = 0; attempt < 2; attempt++) { + try { + const completion = await llmChat({ + messages: [ + { role: "system", content: system }, + { role: "user", content: user }, + ], + max_tokens: opts.maxTokens ?? 4096, + jsonResponse: { type: "json_object" }, + // Use production default retries (2) — real LLMs are non-deterministic. + stream: true, + }); + const raw = completion?.choices?.[0]?.message?.content ?? ""; + return parseModerationResponse( + raw, + messages.map((m) => m.id), + ); + } catch (err) { + lastError = err; + } + } + throw lastError instanceof Error + ? lastError + : new Error(`moderate() failed twice: ${String(lastError)}`); +} + +runIfLLM("LLM E2E — real-model prompt pipeline", () => { + it("clean technical message → no flags, no delete", async () => { + const results = await moderate([ + { + id: "e2e-clean-1", + user: "Budi", + content: "Halo semua, ada yang tau cara setup redis di docker?", + }, + ]); + const r = results[0]; + expect(r.status).toBe("clean"); + expect(r.flags).not.toContain("threat"); + expect(r.recommendedAction).not.toBe("delete"); + }, 120_000); + + it("physics/engineering discussion → clean, no false-positive threat", async () => { + const results = await moderate([ + { + id: "e2e-physics-1", + user: "Scientist", + content: + "Menurutku energi kinetik itu 1/2 mv^2, kalau gravitasi 9.8 m/s^2 di permukaan bumi. Pembahasan teknis aja ya.", + }, + ]); + const r = results[0]; + expect(r.status).toBe("clean"); + expect(r.flags).not.toContain("threat"); + expect(r.flags).not.toContain("violence"); + }, 120_000); + + it("explicit harassment/abuse → flagged high, delete recommended", async () => { + const results = await moderate([ + { + id: "e2e-abuse-1", + user: "Rizky", + content: + "Dasar kampret! Awas saja, gua bakal bunuh lo semua di grup ini, anjing!", + }, + ]); + const r = results[0]; + // Must NOT be clean — either flagged or at least a warn with a flag + expect(["flagged", "warn"]).toContain(r.status); + expect(r.flags.length).toBeGreaterThan(0); + }, 120_000); + + it("offensive username 'Pecinta Pria' + clean content → nickname violation path (warn, NOT delete)", async () => { + const results = await moderate([ + { + id: "e2e-username-1", + user: "Pecinta Pria", + content: "Test message, nothing to see here.", + }, + ]); + const r = results[0]; + // Username-only violation: MUST NOT delete. The whole point of the + // firewall rule — username-only offense → warn/low, never flagged/delete. + expect(r.recommendedAction).not.toBe("delete"); + expect(r.severity).not.toBe("high"); + expect(r.severity).not.toBe("critical"); + }, 120_000); + + it("sexual/provocative username + clean content → offensive_username flag, warn only", async () => { + const results = await moderate([ + { + id: "e2e-username-2", + user: "Cinta", + content: "Pagi semua, ada yang main valo hari ini?", + }, + { + id: "e2e-username-3", + user: "HotBabe", + content: "Biasa aja bro, ngobrol santai.", + }, + ]); + results.forEach((r) => { + // Username-only offense: NEVER delete, never high severity + expect(r.recommendedAction).not.toBe("delete"); + expect(r.severity).not.toBe("high"); + expect(r.severity).not.toBe("critical"); + // If the LLM flags it (not guaranteed for borderline usernames), + // the flag must be username-attributable, not content-level delete. + if (r.flags.includes("offensive_username")) { + expect(r.status).toBe("warn"); + expect(r.recommendedAction).toMatch(/^(none|warn)$/); + } + }); + }, 120_000); + + it("SARA political term ONLY in username + clean content → username warning, NOT content zero-tolerance", async () => { + const results = await moderate([ + { + id: "e2e-username-4", + user: "matikanetanyahu", + content: "OOO GW TAU KARENA APA, tapi gapapa lah", + }, + ]); + const r = results[0]; + // Username-only SARA appearance — must NOT trigger delete (zero-tolerance + // applies to CONTENT). Must be warn/low or at most a content-level low. + expect(r.recommendedAction).not.toBe("delete"); + expect(r.severity).not.toBe("high"); + expect(r.severity).not.toBe("critical"); + }, 120_000); + + it("repeated identical short messages (spam burst) → flagged/warn with spam flag", async () => { + const results = await moderate([ + { + id: "e2e-spam-1", + user: "Spammer", + content: "ok", + repetitions: "5", + }, + ]); + const r = results[0]; + // Single short message with repetitions="5" = the dedup group was 5 — + // the LLM should see it as at least a warning (not clean). + expect(["clean", "flagged", "warn"]).toContain(r.status); + if (r.status !== "clean") { + expect(r.flags.length).toBeGreaterThan(0); + } + }, 120_000); +});