refactor(automod): remove regex classifier — LLM is the sole judge
Build & Deploy (Nix) / build-and-deploy (backend) (push) Successful in 3m2s
Build & Deploy (Nix) / build-and-deploy (discord-gateway) (push) Successful in 2m20s
Build & Deploy (Nix) / build-and-deploy (proxy) (push) Successful in 2m26s
Build & Deploy (Nix) / build-and-deploy (backend) (push) Successful in 3m2s
Build & Deploy (Nix) / build-and-deploy (discord-gateway) (push) Successful in 2m20s
Build & Deploy (Nix) / build-and-deploy (proxy) (push) Successful in 2m26s
Delete fastClassifier.ts (manual regex patterns for phone/email/IP/crypto/ spam/toxicity) and simpleFallback.ts. These hardcoded patterns were the source of false positives (Discord emoji snowflakes matched phone_number, URL digits matched phone, etc.) and produced heuristic verdicts whenever the LLM failed. New flow: Message → LLM (with conversation context, media evidence, user reputation) → verdict. On LLM failure the message is marked 'error' and retried by the recovery worker — no heuristic verdicts, ever. Discord markdown tokens (custom emoji/mentions/timestamps) are normalized to readable placeholders ([emoji:name], @user, @role, #channel, [time]) before reaching the LLM via discordTokens.ts.
This commit is contained in:
@@ -1,20 +1,15 @@
|
||||
/**
|
||||
* ai-analysis-worker.ts
|
||||
*
|
||||
* Two-pass AI moderation analysis worker (Piscina-compatible).
|
||||
* AI moderation analysis worker (Piscina-compatible).
|
||||
*
|
||||
* ## Pipeline
|
||||
*
|
||||
* Message → Layer 1 (fast classifier / heuristic)
|
||||
* │
|
||||
* ├─ clear match → final result (no LLM call)
|
||||
* └─ ambiguous → Layer 2 (LLM evaluator)
|
||||
* Message → LLM evaluator (with conversation context + media evidence)
|
||||
*
|
||||
* Layer 1 runs synchronously in-memory. Layer 2 calls the LLM API via
|
||||
* the existing moderation pipeline (moderationOrchestrator).
|
||||
*
|
||||
* This file replaces the old `aiAnalysisWorker.ts` (archived) with a
|
||||
* simpler, unified worker that combines both layers.
|
||||
* Every message is judged by the LLM — there is no regex/heuristic
|
||||
* pre-classification. A failed LLM call yields an explicit "error" status
|
||||
* (never a heuristic verdict), and the recovery worker retries it later.
|
||||
*/
|
||||
|
||||
import { createChildLogger } from "@/shared/logger/index";
|
||||
@@ -23,8 +18,6 @@ import { initializeDatabase } from "../../shared/database/drizzle.js";
|
||||
import { messageStore } from "../message-capture/messageStore.js";
|
||||
import type { MessageRecord } from "../message-capture/types.js";
|
||||
import { buildConversationContext } from "./conversationContext.js";
|
||||
import { classifyMessage } from "./fastClassifier.js";
|
||||
import type { Layer1Result } from "./fastClassifier.js";
|
||||
import { runModerationAnalysis } from "./moderationOrchestrator.js";
|
||||
|
||||
const logger = createChildLogger("ai-analysis-worker");
|
||||
@@ -89,11 +82,7 @@ export interface MessageBatch {
|
||||
// Worker job types (Piscina entry point)
|
||||
type WorkerJob =
|
||||
| { type: "batch"; conversationKey: string; messages: MessageRecord[] }
|
||||
| {
|
||||
type: "individual";
|
||||
message: MessageRecord;
|
||||
skipNormalAnalysis: boolean;
|
||||
};
|
||||
| { type: "individual"; message: MessageRecord; skipNormalAnalysis: boolean };
|
||||
|
||||
type BatchOkResponse = {
|
||||
ok: true;
|
||||
@@ -186,39 +175,21 @@ export default async function workerRouter(
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Two-pass pipeline
|
||||
// Single-pass LLM pipeline
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Runs the two-pass pipeline on a single message:
|
||||
* 1. Layer 1 — fast heuristic classifier
|
||||
* 2. Layer 2 (if cascade) — LLM-based evaluator
|
||||
* Runs the LLM moderation analysis on a single message.
|
||||
*
|
||||
* Returns the combined AnalysisResult.
|
||||
* The LLM verdict IS the result — confidence, severity, flags and
|
||||
* explanation all come from the model. On failure the message is marked
|
||||
* "error" (explicit, retryable) instead of receiving a heuristic verdict.
|
||||
*/
|
||||
async function runTwoPassPipeline(
|
||||
async function runLLMAnalysis(
|
||||
message: MessageRecord,
|
||||
contextText: string,
|
||||
attachments: Awaited<ReturnType<typeof messageStore.getAttachmentsForMessages>>,
|
||||
): Promise<AnalysisResult> {
|
||||
// ── Layer 1: Fast classifier ──────────────────────────────────────────
|
||||
const layer1Result: Layer1Result = classifyMessage(message);
|
||||
|
||||
logger.debug(
|
||||
{
|
||||
messageId: message.id,
|
||||
layer1Flags: layer1Result.flags,
|
||||
cascade: layer1Result.cascadeToLayer2,
|
||||
},
|
||||
"Layer 1 classification complete",
|
||||
);
|
||||
|
||||
if (!layer1Result.cascadeToLayer2) {
|
||||
// Layer 1 result is final — no LLM call needed
|
||||
return buildResultFromLayer1(message.id, layer1Result);
|
||||
}
|
||||
|
||||
// ── Layer 2: LLM-based evaluation ─────────────────────────────────────
|
||||
try {
|
||||
const moderationResult = await runModerationAnalysis({
|
||||
targets: [message],
|
||||
@@ -231,45 +202,44 @@ async function runTwoPassPipeline(
|
||||
}
|
||||
|
||||
const llmResult = moderationResult.results[0] as unknown as AnalysisResult;
|
||||
return mergeLayers(layer1Result, llmResult);
|
||||
if (llmResult.status === "error") {
|
||||
return llmResult;
|
||||
}
|
||||
|
||||
return {
|
||||
messageId: llmResult.messageId,
|
||||
status: llmResult.status ?? "clean",
|
||||
flags: llmResult.flags ?? [],
|
||||
categories: llmResult.categories ?? [],
|
||||
severity: llmResult.severity ?? "none",
|
||||
confidence: normalizeConfidence(llmResult.confidence),
|
||||
recommendedAction: llmResult.recommendedAction ?? "none",
|
||||
toxicityScore: llmResult.toxicityScore ?? 0,
|
||||
harmScore: llmResult.harmScore ?? 0,
|
||||
jailbreakScore: llmResult.jailbreakScore ?? 0,
|
||||
safetyScore: llmResult.safetyScore ?? 0,
|
||||
explanation:
|
||||
llmResult.explanation?.trim() ||
|
||||
(llmResult.status === "clean"
|
||||
? "Tidak ada indikasi pelanggaran."
|
||||
: "Pesan terindikasi melanggar kebijakan (analisis AI)."),
|
||||
};
|
||||
} catch (error) {
|
||||
const errorMsg = error instanceof Error ? error.message : String(error);
|
||||
logger.warn(
|
||||
{ messageId: message.id, error: errorMsg },
|
||||
"Layer 2 (LLM) analysis failed — falling back to Layer 1 result",
|
||||
"LLM analysis failed for message",
|
||||
);
|
||||
// Fallback to Layer 1 with reduced confidence
|
||||
const fallback = buildResultFromLayer1(message.id, layer1Result);
|
||||
fallback.confidence = Math.min(fallback.confidence, 0.4);
|
||||
return fallback;
|
||||
return buildFallbackResult(message.id, errorMsg);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result builders
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
function buildResultFromLayer1(
|
||||
messageId: string,
|
||||
layer1: Layer1Result,
|
||||
): AnalysisResult {
|
||||
const status = layer1.severity === "none" ? "clean" as const : "flagged" as const;
|
||||
const recommendedAction = mapSeverityToAction(layer1.severity);
|
||||
|
||||
return {
|
||||
messageId,
|
||||
status,
|
||||
flags: layer1.flags,
|
||||
categories: layer1.flags,
|
||||
severity: layer1.severity === "high" ? "high" as const : layer1.severity === "medium" ? "medium" as const : "low" as const,
|
||||
confidence: layer1.confidence,
|
||||
recommendedAction,
|
||||
toxicityScore: layer1.toxicityScore,
|
||||
harmScore: layer1.harmScore,
|
||||
jailbreakScore: 0,
|
||||
safetyScore: 0,
|
||||
explanation: layer1.explanation,
|
||||
};
|
||||
/** Clamp LLM-provided confidence to [0, 1]; default by status when missing. */
|
||||
function normalizeConfidence(raw: number | undefined | null): number {
|
||||
if (typeof raw === "number" && Number.isFinite(raw)) {
|
||||
return Math.min(Math.max(raw, 0), 1);
|
||||
}
|
||||
return 0.7;
|
||||
}
|
||||
|
||||
function buildFallbackResult(
|
||||
@@ -292,70 +262,6 @@ function buildFallbackResult(
|
||||
};
|
||||
}
|
||||
|
||||
function mergeLayers(
|
||||
layer1: Layer1Result,
|
||||
llmResult: AnalysisResult,
|
||||
): AnalysisResult {
|
||||
// Merge flags from both layers (deduplicate)
|
||||
const flagSet = new Set<string>([...layer1.flags, ...(llmResult.flags || [])]);
|
||||
|
||||
// Take the max severity
|
||||
const severityOrder = ["none", "low", "medium", "high", "critical"] as const;
|
||||
const l1Idx = severityOrder.indexOf(layer1.severity);
|
||||
const l2Idx = severityOrder.indexOf(
|
||||
(llmResult.severity ?? "none") as (typeof severityOrder)[number],
|
||||
);
|
||||
const finalSeverity = severityOrder[Math.max(l1Idx, l2Idx)];
|
||||
|
||||
// Combined confidence: weighted average favoring LLM when available
|
||||
const combinedConfidence =
|
||||
0.3 * layer1.confidence + 0.7 * (llmResult.confidence ?? 0.5);
|
||||
|
||||
// Combine scores (take max per dimension)
|
||||
const toxicityScore = Math.max(
|
||||
layer1.toxicityScore,
|
||||
llmResult.toxicityScore ?? 0,
|
||||
);
|
||||
const harmScore = Math.max(layer1.harmScore, llmResult.harmScore ?? 0);
|
||||
|
||||
return {
|
||||
messageId: llmResult.messageId,
|
||||
status: llmResult.status === "error" ? "error" as const : llmResult.status ?? "clean" as const,
|
||||
flags: Array.from(flagSet),
|
||||
categories: [
|
||||
...new Set([
|
||||
...layer1.flags,
|
||||
...(llmResult.categories ?? []),
|
||||
]),
|
||||
],
|
||||
severity: finalSeverity,
|
||||
confidence: Math.min(combinedConfidence, 1),
|
||||
recommendedAction: llmResult.recommendedAction ?? mapSeverityToAction(finalSeverity),
|
||||
toxicityScore,
|
||||
harmScore,
|
||||
jailbreakScore: llmResult.jailbreakScore ?? 0,
|
||||
safetyScore: llmResult.safetyScore ?? 0,
|
||||
explanation: llmResult.explanation ?? layer1.explanation,
|
||||
};
|
||||
}
|
||||
|
||||
function mapSeverityToAction(
|
||||
severity: "none" | "low" | "medium" | "high" | "critical",
|
||||
): AnalysisResult["recommendedAction"] {
|
||||
switch (severity) {
|
||||
case "none":
|
||||
return "none";
|
||||
case "low":
|
||||
return "monitor";
|
||||
case "medium":
|
||||
return "review";
|
||||
case "high":
|
||||
return "delete";
|
||||
case "critical":
|
||||
return "escalate";
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Batch handler
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -391,16 +297,16 @@ async function processBatch(job: {
|
||||
const attachments =
|
||||
await messageStore.getAttachmentsForMessages(allMessageIds);
|
||||
|
||||
// Run two-pass pipeline for each message
|
||||
// Run LLM analysis for each message
|
||||
const analysisResults = await Promise.all(
|
||||
messages.map(async (msg) => {
|
||||
try {
|
||||
return await runTwoPassPipeline(msg, contextText, attachments);
|
||||
return await runLLMAnalysis(msg, contextText, attachments);
|
||||
} catch (error) {
|
||||
const errorMsg = error instanceof Error ? error.message : String(error);
|
||||
logger.error(
|
||||
{ messageId: msg.id, error: errorMsg },
|
||||
"Two-pass pipeline failed for message",
|
||||
"LLM analysis failed for message",
|
||||
);
|
||||
return buildFallbackResult(msg.id, errorMsg);
|
||||
}
|
||||
@@ -435,7 +341,7 @@ async function processBatch(job: {
|
||||
saved: allRows.length,
|
||||
conversationKey,
|
||||
},
|
||||
"Two-pass batch analysis complete",
|
||||
"LLM batch analysis complete",
|
||||
);
|
||||
|
||||
return { ok: true, conversationKey, rows: allRows };
|
||||
@@ -450,22 +356,11 @@ async function processIndividual(job: {
|
||||
message: MessageRecord;
|
||||
skipNormalAnalysis: boolean;
|
||||
}): Promise<IndividualOkResponse | IndividualErrorResponse> {
|
||||
const { message, skipNormalAnalysis } = job;
|
||||
const { message } = job;
|
||||
|
||||
if (skipNormalAnalysis) {
|
||||
// Use Layer 1 only (fast)
|
||||
const layer1Result = classifyMessage(message);
|
||||
|
||||
if (!layer1Result.cascadeToLayer2) {
|
||||
// Layer 1 is sufficient
|
||||
return {
|
||||
ok: true,
|
||||
results: [buildResultFromLayer1(message.id, layer1Result)],
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Full analysis (context + two-pass)
|
||||
// Full analysis (context + LLM). `skipNormalAnalysis` is accepted for
|
||||
// compatibility with the old two-pass protocol but always runs the LLM —
|
||||
// there is no heuristic path anymore.
|
||||
const contextBefore = await messageStore.getConversationContextBefore({
|
||||
channelId: message.channel_id,
|
||||
threadId: message.thread_id,
|
||||
@@ -487,14 +382,14 @@ async function processIndividual(job: {
|
||||
]);
|
||||
|
||||
try {
|
||||
const result = await runTwoPassPipeline(message, contextText, attachments);
|
||||
const result = await runLLMAnalysis(message, contextText, attachments);
|
||||
return { ok: true, results: [result] };
|
||||
} catch (error) {
|
||||
const errorMsg = error instanceof Error ? error.message : String(error);
|
||||
logger.error(
|
||||
{ messageId: message.id, error: errorMsg },
|
||||
"Individual two-pass analysis failed",
|
||||
"Individual LLM analysis failed",
|
||||
);
|
||||
return { ok: true, results: [buildFallbackResult(message.id, errorMsg)] };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ import { createChildLogger } from "@/shared/logger/index";
|
||||
import { encoding_for_model as encodingForModel } from "tiktoken";
|
||||
import { formatMediaEvidenceForPrompt } from "../message-capture/messageMetadata.js";
|
||||
import type { MessageRecord } from "../message-capture/types.js";
|
||||
import { sanitizeDiscordTokens } from "./discordTokens.js";
|
||||
|
||||
const logger = createChildLogger("conversationContext");
|
||||
|
||||
@@ -67,7 +68,9 @@ export function formatMessageForPrompt(
|
||||
msg: MessageRecord,
|
||||
label: "context" | "target",
|
||||
): string {
|
||||
const content = msg.edited_content ?? msg.content;
|
||||
const content = sanitizeDiscordTokens(
|
||||
msg.edited_content ?? msg.content,
|
||||
);
|
||||
const timestamp = formatTimestamp(msg.created_at);
|
||||
const mediaEvidence = formatMediaEvidenceForPrompt(msg.metadata);
|
||||
const mediaSuffix = mediaEvidence ? ` ${mediaEvidence}` : "";
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
/**
|
||||
* discordTokens.ts
|
||||
*
|
||||
* Normalizes Discord markdown tokens (custom emoji, user/role/channel
|
||||
* mentions, timestamps) into readable placeholders before content reaches
|
||||
* the LLM. The numeric snowflake IDs inside these tokens are meaningless to
|
||||
* an LLM and were the source of repeated false positives in the old
|
||||
* regex-based Layer 1 classifier — so raw IDs are never presented.
|
||||
*/
|
||||
|
||||
// <:name:id> | <a:name:id> | <@id> | <@!id> | <@&id> | <#id> | <t:id:style>
|
||||
const DISCORD_TOKEN_RE =
|
||||
/<(?:a?:([^>]{1,32}):(\d{17,20})|@!?(\d{17,20})|@&(\d{17,20})|#(\d{17,20})|t:(\d{10,11})(?::([tTdDRfF]))?)>/g;
|
||||
|
||||
/**
|
||||
* Replaces Discord markdown tokens with readable placeholders.
|
||||
*
|
||||
* - `<:name:id>` / `<a:name:id>` → `[emoji:name]`
|
||||
* - `<@123>` / `<@!123>` → `@user`
|
||||
* - `<@&123>` → `@role`
|
||||
* - `<#123>` → `#channel`
|
||||
* - `<t:123:R>` → `[time]`
|
||||
*
|
||||
* No numeric IDs survive, so digit-shaped patterns can never match them.
|
||||
*/
|
||||
export function sanitizeDiscordTokens(content: string): string {
|
||||
if (!content.includes("<")) return content;
|
||||
return content.replace(
|
||||
DISCORD_TOKEN_RE,
|
||||
(_full, emojiName, _emojiId, _userId, _roleId, _channelId, _time, _style) => {
|
||||
if (emojiName) return `[emoji:${emojiName}]`;
|
||||
// Capture groups tell us which alternative matched by position:
|
||||
// 3=user, 4=role, 5=channel, 6=time
|
||||
if (_userId !== undefined) return "@user";
|
||||
if (_roleId !== undefined) return "@role";
|
||||
if (_channelId !== undefined) return "#channel";
|
||||
if (_time !== undefined) return "[time]";
|
||||
return " ";
|
||||
},
|
||||
);
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
@@ -1,9 +1,12 @@
|
||||
export { startPendingAIAnalysisWorker } from "./aiAnalyzer.js";
|
||||
export { runModerationAnalysis } from "./moderationOrchestrator.js";
|
||||
export { buildSystemPrompt } from "./moderationPrompt.js";
|
||||
export { runSimpleTextFallback } from "./simpleFallback.js";
|
||||
export { sanitizeDiscordTokens } from "./discordTokens.js";
|
||||
|
||||
// ── New two-pass pipeline exports ──────────────────────────────────────────
|
||||
export { classifyMessage } from "./fastClassifier.js";
|
||||
export type { Layer1Result } from "./fastClassifier.js";
|
||||
export type { AnalysisInput, AnalysisResult, WorkerConfig, MessageBatch } from "./ai-analysis-worker.js";
|
||||
// ── Single-pass LLM pipeline exports ─────────────────────────────────────
|
||||
export type {
|
||||
AnalysisInput,
|
||||
AnalysisResult,
|
||||
WorkerConfig,
|
||||
MessageBatch,
|
||||
} from "./ai-analysis-worker.js";
|
||||
|
||||
@@ -82,7 +82,6 @@ async function processIndividualFallback(
|
||||
| { ok: false; results: AnalysisResult[]; error: string };
|
||||
|
||||
let analysisResult: { results: AnalysisResult[] } | null = null;
|
||||
let usedSimpleFallback = false;
|
||||
|
||||
if (workerResult.ok) {
|
||||
const stillIncomplete = workerResult.results.some((r) =>
|
||||
@@ -96,38 +95,12 @@ async function processIndividualFallback(
|
||||
}
|
||||
}
|
||||
|
||||
// Step 2: If normal analysis failed, try SIMPLE fallback via worker
|
||||
if (!analysisResult) {
|
||||
logger.info(
|
||||
{ messageId },
|
||||
"Normal analysis failed -- trying simple text fallback via worker",
|
||||
);
|
||||
|
||||
const simpleResult = (await workerPool.run({
|
||||
type: "individual",
|
||||
message,
|
||||
skipNormalAnalysis: true,
|
||||
} as unknown)) as
|
||||
| { ok: true; results: AnalysisResult[] }
|
||||
| { ok: false; results: AnalysisResult[]; error: string };
|
||||
|
||||
if (simpleResult.ok) {
|
||||
analysisResult = simpleResult;
|
||||
usedSimpleFallback = true;
|
||||
exhaustedOnIncomplete = false;
|
||||
}
|
||||
}
|
||||
|
||||
// No heuristic fallback: an incomplete/errored LLM result stays a
|
||||
// retryable error — the recovery worker picks it up later. Producing a
|
||||
// regex/wordlist verdict here would reintroduce false positives.
|
||||
if (!analysisResult) {
|
||||
throw new Error(
|
||||
`Both normal and simple analysis failed for message ${messageId}`,
|
||||
);
|
||||
}
|
||||
|
||||
if (usedSimpleFallback) {
|
||||
logger.info(
|
||||
{ messageId, status: analysisResult.results[0]?.status },
|
||||
"Used simple text fallback for individual message (via worker)",
|
||||
`LLM analysis failed for message ${messageId}`,
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
|
||||
import { messageStore } from "../message-capture/messageStore.js";
|
||||
import type { MessageRecord } from "../message-capture/types.js";
|
||||
import { sanitizeDiscordTokens } from "./discordTokens.js";
|
||||
|
||||
/** Simple XML-escaping for content text. */
|
||||
export function escapeXml(s: string): string {
|
||||
@@ -29,7 +30,7 @@ export function getAnalysisContent(message: MessageRecord): string {
|
||||
/\[(?:Attachment|Sticker):[^\]]*\]|\[Embed\]/g,
|
||||
"",
|
||||
);
|
||||
return stripped.trim();
|
||||
return sanitizeDiscordTokens(stripped).trim();
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,182 +0,0 @@
|
||||
/**
|
||||
* simpleFallback.ts
|
||||
*
|
||||
* Simple two-step text fallback for cheap/small models.
|
||||
* Step 1: Single-word classification (clean/warn/flagged).
|
||||
* Step 2: Real analysis text (only if not clean).
|
||||
* Extracted from moderationOrchestrator.ts.
|
||||
*/
|
||||
import { createChildLogger } from "@/shared/logger/index";
|
||||
import type {
|
||||
AnalysisResult,
|
||||
MessageRecord,
|
||||
} from "../message-capture/types.js";
|
||||
import { llmChat } from "./llmClient.js";
|
||||
import { getAnalysisContent } from "./moderationBuilders.js";
|
||||
import { sanitizeAiContent } from "./moderationPrompt.js";
|
||||
import { getUserProfile } from "./userProfileStore.js";
|
||||
|
||||
const log = createChildLogger("simpleFallback");
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Simple text-only fallback
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Simple two-step text fallback for cheap/small models.
|
||||
* Step 1: Single-word classification (clean/warn/flagged).
|
||||
* Step 2: Real analysis text (only if not clean).
|
||||
*/
|
||||
export async function runSimpleTextFallback(
|
||||
message: MessageRecord,
|
||||
): Promise<AnalysisResult> {
|
||||
const content = getAnalysisContent(message);
|
||||
const MAX_CONTENT_CHARS = 500;
|
||||
const truncatedContent =
|
||||
content.length > MAX_CONTENT_CHARS
|
||||
? `${content.slice(0, MAX_CONTENT_CHARS)}...`
|
||||
: content;
|
||||
|
||||
let userProfileCtx = "";
|
||||
try {
|
||||
const profile = await getUserProfile(message.user_id);
|
||||
if (profile?.profile_summary) {
|
||||
userProfileCtx = `\n\nProfil pengirim pesan:\n${sanitizeAiContent(profile.profile_summary, 3000, false)}\n`;
|
||||
}
|
||||
} catch {
|
||||
/* non-fatal */
|
||||
}
|
||||
|
||||
// Step 1: Single-word classification
|
||||
const classifyPrompt = `Pesan berikut perlu diklasifikasikan sebagai: clean, warn, atau flagged.
|
||||
|
||||
Aturan:
|
||||
- clean: pesan biasa, percakapan normal, tidak ada pelanggaran
|
||||
- warn: spam ringan, promosi tidak jelas, atau pelanggaran ringan
|
||||
- flagged: harassment, SARA, NSFW, judi, ancaman, atau pelanggaran serius
|
||||
|
||||
PENTING (False Positive Prevention):
|
||||
- Slang Indonesia ("anjay", "wkwk", "njir", "gws", dll) dan makian umum ("asu", "anjing", "bangsat") yang TIDAK ditujukan ke orang lain = clean.
|
||||
- Konten coding/programming (kode, log error, SQL, command line, error message, stack trace, nama library) = clean. JANGAN flag hanya karena ada kata "error" atau "crash" dalam konteks teknis.
|
||||
- Nama proyek, tools, framework (IMPHNEN, Bete, Cursor, Claude, React, Discord) = clean.
|
||||
- Percakapan multilingual (campuran Indonesia-Inggris) = clean.
|
||||
${userProfileCtx}
|
||||
Pesan: "${truncatedContent}"
|
||||
|
||||
Jawab HANYA dengan satu kata: clean, warn, atau flagged`;
|
||||
|
||||
let status: "clean" | "warn" | "flagged";
|
||||
try {
|
||||
const completion = await llmChat({
|
||||
messages: [{ role: "user", content: classifyPrompt }],
|
||||
max_tokens: 10,
|
||||
temperature: 0.1,
|
||||
});
|
||||
const raw =
|
||||
completion?.choices[0]?.message?.content?.trim().toLowerCase() ?? "";
|
||||
if (raw.includes("flagged")) status = "flagged";
|
||||
else if (raw.includes("warn")) status = "warn";
|
||||
else status = "clean";
|
||||
log.info({ messageId: message.id, status, raw }, "Simple fallback step 1");
|
||||
} catch (error) {
|
||||
log.warn(
|
||||
{
|
||||
messageId: message.id,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
},
|
||||
"Simple fallback step 1 failed — defaulting to clean",
|
||||
);
|
||||
status = "clean";
|
||||
}
|
||||
|
||||
// Step 2: Reason + category (only if not clean)
|
||||
let analysis: string;
|
||||
let category = "";
|
||||
|
||||
if (status === "clean") {
|
||||
analysis = `${message.username ?? "user"}: ${content.length > 200 ? `${content.slice(0, 200)}...` : content}. Percakapan normal, tidak ada pelanggaran.`;
|
||||
} else {
|
||||
category = status === "flagged" ? "harassment" : "spam";
|
||||
const categoryOptions =
|
||||
status === "flagged" ? "harassment, gambling, atau sara" : "spam";
|
||||
const reasonPrompt = `Pesan berikut telah diklasifikasikan sebagai "${status}".
|
||||
${userProfileCtx}
|
||||
Pesan: "${truncatedContent}"
|
||||
|
||||
Jelaskan dalam 1-2 kalimat Bahasa Indonesia: APA yang melanggar dan KENAPA. Jangan gunakan kata "mungkin" atau "sepertinya". Jangan tulis ulang pesan. Langsung ke alasan.
|
||||
|
||||
Setelah alasan, sebutkan Kategori: ${categoryOptions}
|
||||
|
||||
Contoh untuk "flagged":
|
||||
Mengandung kata kasar terarah ke individu tertentu sebagai hinaan.
|
||||
Kategori: harassment
|
||||
|
||||
Contoh untuk "flagged":
|
||||
Promosi situs judi online dengan link dan ajakan.
|
||||
Kategori: gambling
|
||||
|
||||
Contoh untuk "warn":
|
||||
Promosi channel Discord tanpa konteks, berpotensi spam.
|
||||
Kategori: spam
|
||||
|
||||
Contoh untuk "warn":
|
||||
Bahasa kasar ringan yang tidak terarah.
|
||||
Kategori: spam`;
|
||||
|
||||
try {
|
||||
const completion = await llmChat({
|
||||
messages: [{ role: "user", content: reasonPrompt }],
|
||||
max_tokens: 80,
|
||||
temperature: 0.3,
|
||||
});
|
||||
analysis = completion?.choices[0]?.message?.content?.trim() ?? "";
|
||||
if (!analysis || analysis.length < 5) {
|
||||
analysis = `Pesan diklasifikasikan sebagai ${status} oleh sistem moderasi otomatis.`;
|
||||
}
|
||||
const categoryMatch = analysis.match(/[Kk]ategori:\s*(\w+)/i);
|
||||
if (categoryMatch) {
|
||||
const parsedCat = categoryMatch[1].toLowerCase();
|
||||
if (["harassment", "spam", "gambling", "sara"].includes(parsedCat))
|
||||
category = parsedCat;
|
||||
analysis = analysis.replace(/[Kk]ategori:\s*\w+\s*/i, "").trim();
|
||||
}
|
||||
log.info(
|
||||
{
|
||||
messageId: message.id,
|
||||
status,
|
||||
category,
|
||||
analysis: analysis.slice(0, 100),
|
||||
},
|
||||
"Simple fallback step 2",
|
||||
);
|
||||
} catch (error) {
|
||||
analysis = `Pesan diklasifikasikan sebagai ${status} oleh sistem moderasi otomatis berdasarkan analisis konten.`;
|
||||
log.warn(
|
||||
{
|
||||
messageId: message.id,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
},
|
||||
"Simple fallback step 2 failed",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
messageId: message.id,
|
||||
status,
|
||||
flags: status === "clean" ? [] : [category],
|
||||
score: status === "flagged" ? 0.7 : status === "warn" ? 0.4 : 0,
|
||||
analysis,
|
||||
categories: status === "clean" ? [] : [category],
|
||||
severity:
|
||||
status === "flagged" ? "medium" : status === "warn" ? "low" : "none",
|
||||
confidence: 0.6,
|
||||
recommendedAction:
|
||||
status === "flagged" ? "review" : status === "warn" ? "warn" : "none",
|
||||
policyVersion: "default-simple-2026-06",
|
||||
evidence:
|
||||
status !== "clean"
|
||||
? [content.length > 120 ? `${content.slice(0, 120)}...` : content]
|
||||
: [],
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user