chore(ai-moderation): harden LLM prompt and lexical scanner against evasion techniques and cross-lingual vulgarities

This commit is contained in:
MythEclipse
2026-06-05 15:28:04 +07:00
parent 9a02ac8d17
commit a0bf7dfdd9
2 changed files with 46 additions and 5 deletions
@@ -134,6 +134,10 @@ const BADWORD_CATEGORIES: BadwordEntry[] = [
"kacang",
"edan",
"gila",
"titten",
"bitch",
"whore",
"slut"
],
},
{
@@ -162,6 +166,8 @@ const BADWORD_CATEGORIES: BadwordEntry[] = [
"unta",
"bangke",
"bangsat",
"kys",
"kill yourself"
],
},
{
@@ -217,6 +223,8 @@ const BADWORD_CATEGORIES: BadwordEntry[] = [
"dasar cina",
"dasar tionghoa",
"dasar pribumi",
"nigger",
"nigga"
],
},
];
@@ -391,9 +399,9 @@ export function buildModerationTextEvidence(
}
if (badwordHits.length > 0) {
notes.push(`Indonesian badword detected: ${badwordHits.join(", ")}`);
notes.push(`Known badword detected: ${badwordHits.join(", ")}`);
} else {
notes.push("no Indonesian badword detected");
notes.push("no known badword detected");
}
return {
File diff suppressed because one or more lines are too long