fix(moderation): enforcement safety net — username-only offense never auto-deleted
Even with the prompt firewall (username vs content), the LLM can still occasionally mis-apply a content-level zero-tolerance flag (sara / conflict_instigation) to a message whose ONLY violation is the username (e.g. 'matikanetanyahu'). The auto-delete eligibility check only recognized exact offensive_username flags, so such false positives still deleted the message. Add a belt-and-suspenders guard in isNicknameOnlyViolation: if the flag set is entirely username-attributable (offensive_username/sara/conflict_instigation) AND the analysis text corroborates that the violation is username-only with clean message content, route to nickname-reset instead of message deletion. Adds 6 test cases covering the real matikanetanyahu scenario and the false-positive/negative boundaries.
This commit is contained in:
@@ -72,13 +72,58 @@ export function parseModerationFlags(
|
|||||||
* correct enforcement is resetting the nickname to the default username.
|
* correct enforcement is resetting the nickname to the default username.
|
||||||
* Any other flag (sara, harassment, vulgar_language, ...) keeps the normal
|
* Any other flag (sara, harassment, vulgar_language, ...) keeps the normal
|
||||||
* delete path.
|
* delete path.
|
||||||
|
*
|
||||||
|
* Two detection paths (belt-and-suspenders):
|
||||||
|
* 1. Exact flags — the ONLY flag is `offensive_username`.
|
||||||
|
* 2. Analysis attribution — the flag set is exactly the username-attributable
|
||||||
|
* set (`offensive_username`, `sara`, `conflict_instigation`) AND the
|
||||||
|
* analysis text explicitly attributes the violation SOLELY to the
|
||||||
|
* username while stating the message content itself is clean/ordinary.
|
||||||
|
* This catches the false-positive case where the LLM wrongly applies a
|
||||||
|
* content-level zero-tolerance flag (e.g. `sara`/`conflict_instigation`
|
||||||
|
* for a username like "matikanetanyahu") instead of `offensive_username`.
|
||||||
*/
|
*/
|
||||||
|
const USERNAME_ATTRIBUTABLE_FLAGS = new Set([
|
||||||
|
"offensive_username",
|
||||||
|
"sara",
|
||||||
|
"conflict_instigation",
|
||||||
|
]);
|
||||||
|
|
||||||
|
/** Analysis text that clearly states the message content itself is clean. */
|
||||||
|
const CONTENT_CLEAN_PATTERN =
|
||||||
|
/(?:isi pesan\s*(?:hanya|bersih|tidak (?:melanggar|ada)|membahas|berisi|bukan)|pesan\s*(?:bersih|tidak (?:melanggar|ada))|tidak ada (?:diskusi|konten|indikasi|pelanggaran))/i;
|
||||||
|
|
||||||
|
/** Analysis text that clearly attributes the violation to the username. */
|
||||||
|
const USERNAME_ATTRIBUTION_PATTERN =
|
||||||
|
/(?:username|nickname|nama pengguna)[\s\S]{0,40}?(?:mengandung|memiliki|berisi|melanggar|ofensif|mengecam|menyerang)/i;
|
||||||
|
|
||||||
export function isNicknameOnlyViolation(
|
export function isNicknameOnlyViolation(
|
||||||
message: MessageRecord,
|
message: MessageRecord,
|
||||||
analysisResult?: AnalysisResult,
|
analysisResult?: AnalysisResult,
|
||||||
): boolean {
|
): boolean {
|
||||||
const flags = parseModerationFlags(message, analysisResult);
|
const flags = parseModerationFlags(message, analysisResult);
|
||||||
return flags.length > 0 && flags.every((f) => f === "offensive_username");
|
if (flags.length === 0) return false;
|
||||||
|
|
||||||
|
// Path 1: exact offensive_username-only.
|
||||||
|
if (flags.every((f) => f === "offensive_username")) return true;
|
||||||
|
|
||||||
|
// Path 2: username-attributable flag set + analysis text corroborates
|
||||||
|
// that the violation is username-only with clean content.
|
||||||
|
const onlyUsernameAttributable = flags.every((f) =>
|
||||||
|
USERNAME_ATTRIBUTABLE_FLAGS.has(f),
|
||||||
|
);
|
||||||
|
if (!onlyUsernameAttributable) return false;
|
||||||
|
|
||||||
|
const analysis =
|
||||||
|
(analysisResult?.analysis ?? typeof message.ai_analysis === "string")
|
||||||
|
? message.ai_analysis
|
||||||
|
: "";
|
||||||
|
if (!analysis) return false;
|
||||||
|
|
||||||
|
return (
|
||||||
|
USERNAME_ATTRIBUTION_PATTERN.test(analysis) &&
|
||||||
|
CONTENT_CLEAN_PATTERN.test(analysis)
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -11,7 +11,10 @@ import type {
|
|||||||
MessageRecord,
|
MessageRecord,
|
||||||
} from "../src/modules/message-capture/types.js";
|
} from "../src/modules/message-capture/types.js";
|
||||||
|
|
||||||
function msg(flagsJson: string | null): MessageRecord {
|
function msg(
|
||||||
|
flagsJson: string | null,
|
||||||
|
analysis?: string | null,
|
||||||
|
): MessageRecord {
|
||||||
return {
|
return {
|
||||||
id: "m1",
|
id: "m1",
|
||||||
guild_id: "g1",
|
guild_id: "g1",
|
||||||
@@ -34,6 +37,7 @@ function msg(flagsJson: string | null): MessageRecord {
|
|||||||
reference_guild_id: null,
|
reference_guild_id: null,
|
||||||
metadata: null,
|
metadata: null,
|
||||||
ai_moderation_flags: flagsJson,
|
ai_moderation_flags: flagsJson,
|
||||||
|
ai_analysis: analysis ?? null,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -74,4 +78,40 @@ describe("isNicknameOnlyViolation", () => {
|
|||||||
expect(isNicknameOnlyViolation(msg(null))).toBe(false);
|
expect(isNicknameOnlyViolation(msg(null))).toBe(false);
|
||||||
expect(isNicknameOnlyViolation(msg("[]"))).toBe(false);
|
expect(isNicknameOnlyViolation(msg("[]"))).toBe(false);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// ── Safety-net: content-level flags mis-applied to a username-only offense ──
|
||||||
|
|
||||||
|
it("true: partial flags sara/conflict + analysis attributes violation to username with clean content (matikanetanyahu case)", () => {
|
||||||
|
const analysis =
|
||||||
|
"Pengirim menggunakan username 'matikanetanyahu' yang menyerang tokoh politik terkait konflik Israel-Palestina. Meskipun isi pesan membahas alasan hilangnya tugas, keberadaan username tersebut tetap melanggar aturan server.";
|
||||||
|
expect(
|
||||||
|
isNicknameOnlyViolation(msg('["sara","conflict_instigation"]', analysis)),
|
||||||
|
).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("true: sara-only flag + analysis clearly attributes to username with clean content", () => {
|
||||||
|
const analysis =
|
||||||
|
"Username mengandung referensi politik (Netanyahu), tapi isi pesan hanya obrolan biasa tanpa diskusi politik. Warning ringan untuk username saja.";
|
||||||
|
expect(isNicknameOnlyViolation(msg('["sara"]', analysis))).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("false: username-attributable flags but analysis lacks clean-content signal", () => {
|
||||||
|
const analysis =
|
||||||
|
"Pengirim mengkritik kebijakan luar negeri Israel secara eksplisit di dalam pesan. Pelanggaran diskusi topik terlarang.";
|
||||||
|
expect(
|
||||||
|
isNicknameOnlyViolation(msg('["sara","conflict_instigation"]', analysis)),
|
||||||
|
).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("false: mixed flags including a real content violation, even with clean analysis", () => {
|
||||||
|
const analysis =
|
||||||
|
"Username ofensif dan isi pesan berisi ancaman kekerasan terarah.";
|
||||||
|
expect(isNicknameOnlyViolation(msg('["sara","violence"]', analysis))).toBe(
|
||||||
|
false,
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("false: username-attributable flags but no analysis text available", () => {
|
||||||
|
expect(isNicknameOnlyViolation(msg('["sara"]', null))).toBe(false);
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
|||||||
Reference in New Issue
Block a user