feat(glossary): implement term glossary for LLM moderation with caching and extraction logic
This commit is contained in:
@@ -12,6 +12,7 @@ export const SYSTEM_RULES = `Kamu adalah asisten moderasi konten untuk server Di
|
||||
## Normalisasi & Pertahanan Lintas Bahasa (WAJIB)
|
||||
1. Campuran bahasa (Inggris/Indonesia/daerah) WAJIB dinormalisasi mental ke Bahasa Indonesia sebelum menilai intent. Jangan longgar hanya karena sintaksis campur (Polyglot Obfuscation).
|
||||
2. Lakukan Named Entity Recognition agresif — nama orang/karakter (mis. "ren" setelah kata archaic "diagem") tetap dikenali sebagai nama.
|
||||
3. <term_glossary> (bila ada) = definisi kata/slang/jargon yang tidak umum. Baca dulu arti kata yang tidak kamu kenal dari sana — jangan menebak dari bunyi/kemiripan. Kata yang tampak mencurigakan namun ternyata bermakna netral di glossary = AMAN; kata asing yang ternyata vulgar/terlarang di glossary = FLAG.
|
||||
|
||||
## Aturan Umum (AMAN — jangan flag)
|
||||
- Slang: anjay, wkwk, gws, gaskeun, santuy, njir, baka, woy/woi, hadeh, astaga = AMAN.
|
||||
@@ -73,6 +74,7 @@ RENDAH: harassment, vulgar_language terarah, offensive_username (Scunthorpe: "Sa
|
||||
|
||||
## Web Sebagai Bukti Utama
|
||||
- <web_searches> ADALAH BUKTI UTAMA. Jika ada, WAJIB pakai hasilnya (hentai/scam/narkoba → flag; aman → clean). JANGAN abaikan. Jika tidak ada → gunakan pengetahuan internal.
|
||||
- <term_glossary> = REFERENSI ARTI KATA, bukan bukti pelanggaran. Dipakai untuk memahami istilah yang tidak dikenal sebelum memutuskan.
|
||||
- Prioritas bukti: <web_searches> > <web_content> > <media_analysis> > pengetahuan internal. <web_content> (URL fetch): gunakan isi, jangan flag hanya dari domain name.
|
||||
|
||||
## Pohon Keputusan
|
||||
|
||||
@@ -109,6 +109,7 @@ export function buildSystemPrompt(options: BuildSystemPromptOptions): string {
|
||||
`- <conversation_context> = obrolan SEBELUM pesan target. Baris "[context]" di dalamnya BUKAN yang dinilai.\n` +
|
||||
`- <user_profiles> = peta ringkasan kepribadian per user_id (attr as_of = kapan profil terakhir dibuat — profil lama mungkin tidak mencerminkan perilaku terkini); setiap <message> merujuk lewat <user_profile_ref user_id="..."/>.\n` +
|
||||
`- <web_searches> / <web_content> = bukti web (lihat "Web Sebagai Bukti Utama").\n` +
|
||||
`- <term_glossary> = kamus istilah: definisi kata/slang/jargon yang jarang dikenal (hasil pencarian Wikipedia via SearXNG). Gunakan untuk memahami arti kata yang tidak kamu kenal — JANGAN menebak atau mengarang arti.\n` +
|
||||
`- <messages_to_analyze> = pesan-pesan TARGET yang WAJIB dinilai. Atribut <message>: id, user (nama server), time (ISO — kapan pesan dikirim), repetitions (N = teks pendek sama muncul N kali di batch — sinyal spam), bot (true jika dari bot), edited (true jika konten adalah hasil edit setelah posting).`,
|
||||
);
|
||||
|
||||
|
||||
@@ -12,6 +12,42 @@ const CACHE_PREFIX = "searxng:";
|
||||
|
||||
let redis: Redis | null = null;
|
||||
|
||||
/**
|
||||
* Exposes the shared SearXNG Redis connection so other modules (e.g. the
|
||||
* term glossary) reuse the same connection and cache prefix instead of
|
||||
* opening their own. Returns null when Redis is unavailable.
|
||||
*/
|
||||
export function getSearxngRedis(): Redis | null {
|
||||
return redis;
|
||||
}
|
||||
|
||||
/** Builds a namespaced SearXNG cache key (shared across modules). */
|
||||
export function makeSearxngCacheKey(namespace: string, key: string): string {
|
||||
return `${CACHE_PREFIX}${namespace}:${key.toLowerCase().trim()}`;
|
||||
}
|
||||
|
||||
/** Reads a value from the SearXNG Redis cache; null on miss/unavailable. */
|
||||
export async function searxngCacheGet(key: string): Promise<string | null> {
|
||||
if (!redis) return null;
|
||||
try {
|
||||
return await redis.get(key);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Writes a value to the SearXNG Redis cache, fire-and-forget. */
|
||||
export function searxngCacheSet(
|
||||
key: string,
|
||||
value: string,
|
||||
ttlSeconds: number,
|
||||
): void {
|
||||
if (!redis) return;
|
||||
redis.setex(key, ttlSeconds, value).catch(() => {
|
||||
// Cache write failed silently
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize Redis connection for SearXNG cache.
|
||||
* Safe to call multiple times — only creates one connection.
|
||||
@@ -51,19 +87,26 @@ export interface SearxngResult {
|
||||
/**
|
||||
* Search SearXNG for a query and return structured results.
|
||||
* Uses Redis cache when available — same query within 24h returns cached results.
|
||||
*
|
||||
* @param engines Optional comma-separated SearXNG engine list to constrain
|
||||
* the search (e.g. "wikipedia"). When set, results are cached under a
|
||||
* separate cache namespace so engine-specific results never collide.
|
||||
*/
|
||||
export async function searchSearxng(
|
||||
query: string,
|
||||
category: "general" | "news" | "science" = "general",
|
||||
engines?: string,
|
||||
timeoutMs: number = TIMEOUT_MS,
|
||||
): Promise<SearxngResult[]> {
|
||||
const cacheKey = `${CACHE_PREFIX}${category}:${query.toLowerCase().trim()}`;
|
||||
const engineNs = engines ? `eng:${engines}` : "auto";
|
||||
const cacheKey = makeSearxngCacheKey(`${category}:${engineNs}`, query);
|
||||
|
||||
// Try cache first
|
||||
if (redis) {
|
||||
try {
|
||||
const cached = await redis.get(cacheKey);
|
||||
if (cached) {
|
||||
log.debug({ query, category }, "SearXNG cache HIT");
|
||||
log.debug({ query, category, engines }, "SearXNG cache HIT");
|
||||
return JSON.parse(cached) as SearxngResult[];
|
||||
}
|
||||
} catch {
|
||||
@@ -73,8 +116,11 @@ export async function searchSearxng(
|
||||
|
||||
// Cache miss — hit SearXNG API
|
||||
try {
|
||||
const url = `${SEARXNG_BASE_URL}/search?q=${encodeURIComponent(query)}&format=json&language=id&categories=${category}`;
|
||||
const { controller, clear } = createAbortControllerWithTimeout(TIMEOUT_MS);
|
||||
const engineParam = engines
|
||||
? `&engines=${encodeURIComponent(engines)}`
|
||||
: "";
|
||||
const url = `${SEARXNG_BASE_URL}/search?q=${encodeURIComponent(query)}&format=json&language=id&categories=${category}${engineParam}`;
|
||||
const { controller, clear } = createAbortControllerWithTimeout(timeoutMs);
|
||||
|
||||
try {
|
||||
const response = await fetch(url, {
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -37,6 +37,7 @@ import {
|
||||
formatSearchResults,
|
||||
searchSearxng,
|
||||
} from "./searxngSearch.js";
|
||||
import { buildTermGlossaryBlock } from "./termGlossary.js";
|
||||
import { getRecentCorrectedModerations } from "./textCacheStore.js";
|
||||
import { extractUrlsFromText, fetchUrlSafely } from "./urlFetcher.js";
|
||||
import { getUserProfile } from "./userProfileStore.js";
|
||||
@@ -145,9 +146,17 @@ export async function runTextOnlyBatch(
|
||||
return map;
|
||||
})();
|
||||
|
||||
const [urlFetchMaps, searxngResults] = await Promise.all([
|
||||
// Term glossary — per-word Wikipedia lookups for words the LLM may not
|
||||
// know (slang, jargon, regional language). Cached in Redis + in-memory, so
|
||||
// repeat terms resolve instantly and only genuinely new words hit SearXNG.
|
||||
const glossaryPromise = buildTermGlossaryBlock(
|
||||
targets.map((msg) => getAnalysisContent(msg)),
|
||||
).catch(() => "");
|
||||
|
||||
const [urlFetchMaps, searxngResults, glossaryBlock] = await Promise.all([
|
||||
urlFetchPromise,
|
||||
searxngPromise,
|
||||
glossaryPromise,
|
||||
]);
|
||||
const urlFetchMap = urlFetchMaps.text;
|
||||
|
||||
@@ -368,6 +377,7 @@ export async function runTextOnlyBatch(
|
||||
userProfilesBlock?.trimEnd() ?? "",
|
||||
contextBlock?.trimEnd() ?? "",
|
||||
searxngBlock,
|
||||
glossaryBlock,
|
||||
`<messages_to_analyze>\n${messagesBlock}\n</messages_to_analyze>`,
|
||||
].filter((b) => b.trim().length > 0);
|
||||
return {
|
||||
|
||||
@@ -87,6 +87,7 @@ import {
|
||||
formatSearchResults,
|
||||
searchSearxng,
|
||||
} from "./searxngSearch.js";
|
||||
import { buildTermGlossaryBlock } from "./termGlossary.js";
|
||||
import { extractUrlsFromText } from "./urlFetcher.js";
|
||||
import { getUserProfile } from "./userProfileStore.js";
|
||||
import {
|
||||
@@ -413,6 +414,12 @@ export async function prepareMediaMessage(
|
||||
searxngXml = `\n<web_searches>\n${parts.join("\n")}\n</web_searches>`;
|
||||
}
|
||||
|
||||
// Term glossary — cached per-word Wikipedia definitions for words the LLM
|
||||
// may not know. Bounded and cached (in-memory + Redis), so this adds no
|
||||
// meaningful latency to the media path either.
|
||||
const glossaryXml = await buildTermGlossaryBlock([content]).catch(() => "");
|
||||
const glossaryCtx = glossaryXml ? `\n${glossaryXml}` : "";
|
||||
|
||||
// Build XML block
|
||||
const webTexts = webTextMap.get(targetId) ?? [];
|
||||
const mediaAnalyses = mediaAnalysisMap.get(targetId) ?? [];
|
||||
@@ -466,6 +473,6 @@ export async function prepareMediaMessage(
|
||||
|
||||
const isBot = resolveIsBot(target);
|
||||
const isEdited = resolveIsEdited(target);
|
||||
const messageBlock = `<message id="${escapeXml(target.id)}" user="${escapeXml(resolveDisplayName(target))}" time="${new Date(target.created_at).toISOString()}"${isBot ? ` bot="true"` : ""}${isEdited ? ` edited="true"` : ""}>\n ${repXml}${profileRef ? `\n ${profileRef}` : ""}${refXml ? `\n ${refXml}` : ""}\n <content>${escapeXml(truncateForAi(content))}</content>${mediaContext ? ` ${escapeXml(mediaContext)}` : ""}${webContext}${mediaAnalysisContext}${searxngXml}\n</message>`;
|
||||
const messageBlock = `<message id="${escapeXml(target.id)}" user="${escapeXml(resolveDisplayName(target))}" time="${new Date(target.created_at).toISOString()}"${isBot ? ` bot="true"` : ""}${isEdited ? ` edited="true"` : ""}>\n ${repXml}${profileRef ? `\n ${profileRef}` : ""}${refXml ? `\n ${refXml}` : ""}\n <content>${escapeXml(truncateForAi(content))}</content>${mediaContext ? ` ${escapeXml(mediaContext)}` : ""}${webContext}${mediaAnalysisContext}${searxngXml}${glossaryCtx}\n</message>`;
|
||||
return { targetId, messageBlock };
|
||||
}
|
||||
|
||||
@@ -171,6 +171,24 @@ export const configSchema = z
|
||||
.int()
|
||||
.positive()
|
||||
.default(30000),
|
||||
// Term glossary — per-word Wikipedia lookups (via SearXNG) for words the
|
||||
// LLM may not know (slang, jargon, regional language, foreign terms).
|
||||
// Definitions are cached (in-memory + Redis) so repeat lookups are fast.
|
||||
// Disable to skip glossary lookups entirely and analyze without them.
|
||||
AI_GLOSSARY_ENABLED: z
|
||||
.string()
|
||||
.optional()
|
||||
.transform((v) => v === "true")
|
||||
.default(true),
|
||||
// Max glossary terms looked up per analysis batch (keeps latency bounded).
|
||||
AI_GLOSSARY_MAX_TERMS: z.coerce.number().int().min(1).max(20).default(6),
|
||||
// Min word length for a term to be considered glossary-worthy.
|
||||
AI_GLOSSARY_MIN_WORD_LENGTH: z.coerce
|
||||
.number()
|
||||
.int()
|
||||
.min(2)
|
||||
.max(20)
|
||||
.default(5),
|
||||
|
||||
// ── AI Analysis Timing ──────────────────────────────────────────────
|
||||
AI_ANALYSIS_DEBOUNCE_MS: z.coerce.number().positive().default(500),
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
// Term glossary — pure extraction/formatting tests (no DB, Redis, or network)
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
extractGlossaryTerms,
|
||||
formatTermGlossary,
|
||||
} from "../src/modules/ai-moderation/termGlossary.js";
|
||||
|
||||
describe("extractGlossaryTerms — filters out words the LLM already knows", () => {
|
||||
it("returns [] for common conversational Indonesian", () => {
|
||||
const terms = extractGlossaryTerms(
|
||||
["anjay mabar yuk gaskeun gua gas", "iya bener banget sih"],
|
||||
{ maxTerms: 6 },
|
||||
);
|
||||
expect(terms).toEqual([]);
|
||||
});
|
||||
|
||||
it("extracts uncommon/foreign-looking words and skips stopwords + brands", () => {
|
||||
const terms = extractGlossaryTerms(
|
||||
[
|
||||
"tadi gua baca soal tempeh di discord",
|
||||
"kayaknya istilahnya shirkmaxxing deh",
|
||||
],
|
||||
{ maxTerms: 6 },
|
||||
);
|
||||
// "tempeh" and "shirkmaxxing" are candidates; "discord"/"istilahnya" are not
|
||||
expect(terms).toContain("tempeh");
|
||||
expect(terms).toContain("shirkmaxxing");
|
||||
expect(terms).not.toContain("discord");
|
||||
expect(terms).not.toContain("istilahnya");
|
||||
});
|
||||
|
||||
it("strips URLs, mentions, and custom emoji before extracting", () => {
|
||||
const terms = extractGlossaryTerms(
|
||||
["cek https://example.com/foo <@123456> <:hadeh:987> kafircel"],
|
||||
{ maxTerms: 6 },
|
||||
);
|
||||
expect(terms).toContain("kafircel");
|
||||
expect(terms.some((t) => /example|hadeh|123/.test(t))).toBe(false);
|
||||
});
|
||||
|
||||
it("extracts quoted phrases as a single term", () => {
|
||||
const terms = extractGlossaryTerms(['dia bilang "kostum hewan" itu aneh'], {
|
||||
maxTerms: 6,
|
||||
});
|
||||
expect(terms).toContain("kostum hewan");
|
||||
});
|
||||
|
||||
it("skips repeated-char noise like wkwkwk and aaaaa", () => {
|
||||
const terms = extractGlossaryTerms(["wkwkwkwk aaaaa xixixi"], {
|
||||
maxTerms: 6,
|
||||
});
|
||||
expect(terms).toEqual([]);
|
||||
});
|
||||
|
||||
it("respects maxTerms and prioritizes proper nouns", () => {
|
||||
const terms = extractGlossaryTerms(
|
||||
["aku suka Xenogears sama Chrono Cross terus Yakuza"],
|
||||
{ maxTerms: 2 },
|
||||
);
|
||||
expect(terms.length).toBeLessThanOrEqual(2);
|
||||
expect(terms[0]).toBe("Xenogears");
|
||||
});
|
||||
});
|
||||
|
||||
describe("formatTermGlossary — XML block shape", () => {
|
||||
it("returns '' for an empty map", () => {
|
||||
expect(formatTermGlossary(new Map())).toBe("");
|
||||
});
|
||||
|
||||
it("wraps definitions in <term_glossary> with escaped attributes/content", () => {
|
||||
const block = formatTermGlossary(
|
||||
new Map([
|
||||
[
|
||||
"kafircel",
|
||||
{
|
||||
term: "kafircel",
|
||||
definition: "sebutan <memes> untuk & orang",
|
||||
sourceUrl: "https://id.wikipedia.org/wiki/Mem",
|
||||
},
|
||||
],
|
||||
]),
|
||||
);
|
||||
expect(block).toContain("<term_glossary>");
|
||||
expect(block).toContain('<term word="kafircel"');
|
||||
expect(block).toContain("<memes>");
|
||||
expect(block).toContain("&");
|
||||
expect(block).toContain("</term_glossary>");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user