fix(moderation): enforcement safety net — username-only offense never auto-deleted

Even with the prompt firewall (username vs content), the LLM can still
occasionally mis-apply a content-level zero-tolerance flag (sara /
conflict_instigation) to a message whose ONLY violation is the username
(e.g. 'matikanetanyahu'). The auto-delete eligibility check only recognized
exact offensive_username flags, so such false positives still deleted the
message.

Add a belt-and-suspenders guard in isNicknameOnlyViolation: if the flag set
is entirely username-attributable (offensive_username/sara/conflict_instigation)
AND the analysis text corroborates that the violation is username-only with
clean message content, route to nickname-reset instead of message deletion.

Adds 6 test cases covering the real matikanetanyahu scenario and the
false-positive/negative boundaries.
This commit is contained in:
asepharyana
2026-09-03 20:39:17 +07:00
parent 68fbaa631a
commit 85204ca6f0
2 changed files with 87 additions and 2 deletions
@@ -72,13 +72,58 @@ export function parseModerationFlags(
* correct enforcement is resetting the nickname to the default username. * correct enforcement is resetting the nickname to the default username.
* Any other flag (sara, harassment, vulgar_language, ...) keeps the normal * Any other flag (sara, harassment, vulgar_language, ...) keeps the normal
* delete path. * delete path.
*
* Two detection paths (belt-and-suspenders):
* 1. Exact flags — the ONLY flag is `offensive_username`.
* 2. Analysis attribution — the flag set is exactly the username-attributable
* set (`offensive_username`, `sara`, `conflict_instigation`) AND the
* analysis text explicitly attributes the violation SOLELY to the
* username while stating the message content itself is clean/ordinary.
* This catches the false-positive case where the LLM wrongly applies a
* content-level zero-tolerance flag (e.g. `sara`/`conflict_instigation`
* for a username like "matikanetanyahu") instead of `offensive_username`.
*/ */
const USERNAME_ATTRIBUTABLE_FLAGS = new Set([
"offensive_username",
"sara",
"conflict_instigation",
]);
/** Analysis text that clearly states the message content itself is clean. */
const CONTENT_CLEAN_PATTERN =
/(?:isi pesan\s*(?:hanya|bersih|tidak (?:melanggar|ada)|membahas|berisi|bukan)|pesan\s*(?:bersih|tidak (?:melanggar|ada))|tidak ada (?:diskusi|konten|indikasi|pelanggaran))/i;
/** Analysis text that clearly attributes the violation to the username. */
const USERNAME_ATTRIBUTION_PATTERN =
/(?:username|nickname|nama pengguna)[\s\S]{0,40}?(?:mengandung|memiliki|berisi|melanggar|ofensif|mengecam|menyerang)/i;
export function isNicknameOnlyViolation( export function isNicknameOnlyViolation(
message: MessageRecord, message: MessageRecord,
analysisResult?: AnalysisResult, analysisResult?: AnalysisResult,
): boolean { ): boolean {
const flags = parseModerationFlags(message, analysisResult); const flags = parseModerationFlags(message, analysisResult);
return flags.length > 0 && flags.every((f) => f === "offensive_username"); if (flags.length === 0) return false;
// Path 1: exact offensive_username-only.
if (flags.every((f) => f === "offensive_username")) return true;
// Path 2: username-attributable flag set + analysis text corroborates
// that the violation is username-only with clean content.
const onlyUsernameAttributable = flags.every((f) =>
USERNAME_ATTRIBUTABLE_FLAGS.has(f),
);
if (!onlyUsernameAttributable) return false;
const analysis =
(analysisResult?.analysis ?? typeof message.ai_analysis === "string")
? message.ai_analysis
: "";
if (!analysis) return false;
return (
USERNAME_ATTRIBUTION_PATTERN.test(analysis) &&
CONTENT_CLEAN_PATTERN.test(analysis)
);
} }
/** /**
@@ -11,7 +11,10 @@ import type {
MessageRecord, MessageRecord,
} from "../src/modules/message-capture/types.js"; } from "../src/modules/message-capture/types.js";
function msg(flagsJson: string | null): MessageRecord { function msg(
flagsJson: string | null,
analysis?: string | null,
): MessageRecord {
return { return {
id: "m1", id: "m1",
guild_id: "g1", guild_id: "g1",
@@ -34,6 +37,7 @@ function msg(flagsJson: string | null): MessageRecord {
reference_guild_id: null, reference_guild_id: null,
metadata: null, metadata: null,
ai_moderation_flags: flagsJson, ai_moderation_flags: flagsJson,
ai_analysis: analysis ?? null,
}; };
} }
@@ -74,4 +78,40 @@ describe("isNicknameOnlyViolation", () => {
expect(isNicknameOnlyViolation(msg(null))).toBe(false); expect(isNicknameOnlyViolation(msg(null))).toBe(false);
expect(isNicknameOnlyViolation(msg("[]"))).toBe(false); expect(isNicknameOnlyViolation(msg("[]"))).toBe(false);
}); });
// ── Safety-net: content-level flags mis-applied to a username-only offense ──
it("true: partial flags sara/conflict + analysis attributes violation to username with clean content (matikanetanyahu case)", () => {
const analysis =
"Pengirim menggunakan username 'matikanetanyahu' yang menyerang tokoh politik terkait konflik Israel-Palestina. Meskipun isi pesan membahas alasan hilangnya tugas, keberadaan username tersebut tetap melanggar aturan server.";
expect(
isNicknameOnlyViolation(msg('["sara","conflict_instigation"]', analysis)),
).toBe(true);
});
it("true: sara-only flag + analysis clearly attributes to username with clean content", () => {
const analysis =
"Username mengandung referensi politik (Netanyahu), tapi isi pesan hanya obrolan biasa tanpa diskusi politik. Warning ringan untuk username saja.";
expect(isNicknameOnlyViolation(msg('["sara"]', analysis))).toBe(true);
});
it("false: username-attributable flags but analysis lacks clean-content signal", () => {
const analysis =
"Pengirim mengkritik kebijakan luar negeri Israel secara eksplisit di dalam pesan. Pelanggaran diskusi topik terlarang.";
expect(
isNicknameOnlyViolation(msg('["sara","conflict_instigation"]', analysis)),
).toBe(false);
});
it("false: mixed flags including a real content violation, even with clean analysis", () => {
const analysis =
"Username ofensif dan isi pesan berisi ancaman kekerasan terarah.";
expect(isNicknameOnlyViolation(msg('["sara","violence"]', analysis))).toBe(
false,
);
});
it("false: username-attributable flags but no analysis text available", () => {
expect(isNicknameOnlyViolation(msg('["sara"]', null))).toBe(false);
});
}); });