fix(moderation): enforcement safety net — username-only offense never auto-deleted
Even with the prompt firewall (username vs content), the LLM can still occasionally mis-apply a content-level zero-tolerance flag (sara / conflict_instigation) to a message whose ONLY violation is the username (e.g. 'matikanetanyahu'). The auto-delete eligibility check only recognized exact offensive_username flags, so such false positives still deleted the message. Add a belt-and-suspenders guard in isNicknameOnlyViolation: if the flag set is entirely username-attributable (offensive_username/sara/conflict_instigation) AND the analysis text corroborates that the violation is username-only with clean message content, route to nickname-reset instead of message deletion. Adds 6 test cases covering the real matikanetanyahu scenario and the false-positive/negative boundaries.
This commit is contained in:
@@ -72,13 +72,58 @@ export function parseModerationFlags(
|
||||
* correct enforcement is resetting the nickname to the default username.
|
||||
* Any other flag (sara, harassment, vulgar_language, ...) keeps the normal
|
||||
* delete path.
|
||||
*
|
||||
* Two detection paths (belt-and-suspenders):
|
||||
* 1. Exact flags — the ONLY flag is `offensive_username`.
|
||||
* 2. Analysis attribution — the flag set is exactly the username-attributable
|
||||
* set (`offensive_username`, `sara`, `conflict_instigation`) AND the
|
||||
* analysis text explicitly attributes the violation SOLELY to the
|
||||
* username while stating the message content itself is clean/ordinary.
|
||||
* This catches the false-positive case where the LLM wrongly applies a
|
||||
* content-level zero-tolerance flag (e.g. `sara`/`conflict_instigation`
|
||||
* for a username like "matikanetanyahu") instead of `offensive_username`.
|
||||
*/
|
||||
const USERNAME_ATTRIBUTABLE_FLAGS = new Set([
|
||||
"offensive_username",
|
||||
"sara",
|
||||
"conflict_instigation",
|
||||
]);
|
||||
|
||||
/** Analysis text that clearly states the message content itself is clean. */
|
||||
const CONTENT_CLEAN_PATTERN =
|
||||
/(?:isi pesan\s*(?:hanya|bersih|tidak (?:melanggar|ada)|membahas|berisi|bukan)|pesan\s*(?:bersih|tidak (?:melanggar|ada))|tidak ada (?:diskusi|konten|indikasi|pelanggaran))/i;
|
||||
|
||||
/** Analysis text that clearly attributes the violation to the username. */
|
||||
const USERNAME_ATTRIBUTION_PATTERN =
|
||||
/(?:username|nickname|nama pengguna)[\s\S]{0,40}?(?:mengandung|memiliki|berisi|melanggar|ofensif|mengecam|menyerang)/i;
|
||||
|
||||
export function isNicknameOnlyViolation(
|
||||
message: MessageRecord,
|
||||
analysisResult?: AnalysisResult,
|
||||
): boolean {
|
||||
const flags = parseModerationFlags(message, analysisResult);
|
||||
return flags.length > 0 && flags.every((f) => f === "offensive_username");
|
||||
if (flags.length === 0) return false;
|
||||
|
||||
// Path 1: exact offensive_username-only.
|
||||
if (flags.every((f) => f === "offensive_username")) return true;
|
||||
|
||||
// Path 2: username-attributable flag set + analysis text corroborates
|
||||
// that the violation is username-only with clean content.
|
||||
const onlyUsernameAttributable = flags.every((f) =>
|
||||
USERNAME_ATTRIBUTABLE_FLAGS.has(f),
|
||||
);
|
||||
if (!onlyUsernameAttributable) return false;
|
||||
|
||||
const analysis =
|
||||
(analysisResult?.analysis ?? typeof message.ai_analysis === "string")
|
||||
? message.ai_analysis
|
||||
: "";
|
||||
if (!analysis) return false;
|
||||
|
||||
return (
|
||||
USERNAME_ATTRIBUTION_PATTERN.test(analysis) &&
|
||||
CONTENT_CLEAN_PATTERN.test(analysis)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -11,7 +11,10 @@ import type {
|
||||
MessageRecord,
|
||||
} from "../src/modules/message-capture/types.js";
|
||||
|
||||
function msg(flagsJson: string | null): MessageRecord {
|
||||
function msg(
|
||||
flagsJson: string | null,
|
||||
analysis?: string | null,
|
||||
): MessageRecord {
|
||||
return {
|
||||
id: "m1",
|
||||
guild_id: "g1",
|
||||
@@ -34,6 +37,7 @@ function msg(flagsJson: string | null): MessageRecord {
|
||||
reference_guild_id: null,
|
||||
metadata: null,
|
||||
ai_moderation_flags: flagsJson,
|
||||
ai_analysis: analysis ?? null,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -74,4 +78,40 @@ describe("isNicknameOnlyViolation", () => {
|
||||
expect(isNicknameOnlyViolation(msg(null))).toBe(false);
|
||||
expect(isNicknameOnlyViolation(msg("[]"))).toBe(false);
|
||||
});
|
||||
|
||||
// ── Safety-net: content-level flags mis-applied to a username-only offense ──
|
||||
|
||||
it("true: partial flags sara/conflict + analysis attributes violation to username with clean content (matikanetanyahu case)", () => {
|
||||
const analysis =
|
||||
"Pengirim menggunakan username 'matikanetanyahu' yang menyerang tokoh politik terkait konflik Israel-Palestina. Meskipun isi pesan membahas alasan hilangnya tugas, keberadaan username tersebut tetap melanggar aturan server.";
|
||||
expect(
|
||||
isNicknameOnlyViolation(msg('["sara","conflict_instigation"]', analysis)),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it("true: sara-only flag + analysis clearly attributes to username with clean content", () => {
|
||||
const analysis =
|
||||
"Username mengandung referensi politik (Netanyahu), tapi isi pesan hanya obrolan biasa tanpa diskusi politik. Warning ringan untuk username saja.";
|
||||
expect(isNicknameOnlyViolation(msg('["sara"]', analysis))).toBe(true);
|
||||
});
|
||||
|
||||
it("false: username-attributable flags but analysis lacks clean-content signal", () => {
|
||||
const analysis =
|
||||
"Pengirim mengkritik kebijakan luar negeri Israel secara eksplisit di dalam pesan. Pelanggaran diskusi topik terlarang.";
|
||||
expect(
|
||||
isNicknameOnlyViolation(msg('["sara","conflict_instigation"]', analysis)),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it("false: mixed flags including a real content violation, even with clean analysis", () => {
|
||||
const analysis =
|
||||
"Username ofensif dan isi pesan berisi ancaman kekerasan terarah.";
|
||||
expect(isNicknameOnlyViolation(msg('["sara","violence"]', analysis))).toBe(
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
it("false: username-attributable flags but no analysis text available", () => {
|
||||
expect(isNicknameOnlyViolation(msg('["sara"]', null))).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user