fix(ai-moderation): never cache vision outputs that claim 'no image seen'
Root cause (3rd layer after50371bd+4f4c435): a vision model run (2026-08-10) returned 'Maaf, saya tidak melihat gambar apapun yang terlampir...' and that text was cached as a VALID vision_llm result (image + phash keys, 24h/7d TTL). Every subsequent analysis of the same image (same hash/phash) hit the poisoned cache, so image analysis looked broken forever even though 9router responded fine — the moderation LLM wrote 'lampiran yang gagal terbaca' from a cache hit. Also: mimo via 9router streams reasoning in delta.reasoning + delta.reasoning_details[].text (content:"") — extractChunkText only read delta.reasoning_content, so those runs aggregated empty → 'Vision API null response' (observed 08:54/09:07/09:38). Fixes: - llmClient.extractChunkText: fall back to delta.reasoning and reasoning_details[].text (mimo), on top of reasoning_content (gemma). - visionAnalyzer: isNoImageSeenText() detects 'no image' style outputs; such results are NEVER cached, and poisoned entries are purged when hit (LRU/DB/phash) so re-analysis actually re-runs vision. - Tests: reasoning/reasoning_details extraction + isNoImageSeenText (Indonesian + English, no false positives on real descriptions).
This commit is contained in:
@@ -56,7 +56,16 @@ export async function withLlmConcurrency<T>(fn: () => Promise<T>): Promise<T> {
|
|||||||
*/
|
*/
|
||||||
type LLMResponseChunk = {
|
type LLMResponseChunk = {
|
||||||
choices?: Array<{
|
choices?: Array<{
|
||||||
delta?: { content?: string | null; reasoning_content?: string | null };
|
delta?: {
|
||||||
|
content?: string | null;
|
||||||
|
reasoning_content?: string | null;
|
||||||
|
reasoning?: string | null;
|
||||||
|
reasoning_details?: Array<{
|
||||||
|
type?: string;
|
||||||
|
text?: string;
|
||||||
|
index?: number;
|
||||||
|
}> | null;
|
||||||
|
};
|
||||||
message?: { content?: string | null };
|
message?: { content?: string | null };
|
||||||
finish_reason?: string | null;
|
finish_reason?: string | null;
|
||||||
text?: string;
|
text?: string;
|
||||||
@@ -69,18 +78,28 @@ type LLMResponseChunk = {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Extract the textual payload from a single streaming chunk. Prefers
|
* Extract the textual payload from a single streaming chunk. Prefers
|
||||||
* `delta.content`; falls back to `delta.reasoning_content` (DeepSeek-style /
|
* `delta.content`; falls back to reasoning fields so reasoning-only models
|
||||||
* Cloudflare gemma stream ALL output there with content:"") so reasoning-only
|
* still produce usable aggregated text. Providers differ in the field name:
|
||||||
* models still produce usable aggregated text. Exported for unit tests.
|
* - DeepSeek-style / Cloudflare gemma → `delta.reasoning_content`
|
||||||
|
* - mimo (via 9router) streams reasoning in `delta.reasoning` +
|
||||||
|
* `delta.reasoning_details[].text` (content:"") — without these fallbacks
|
||||||
|
* vision aggregation came back empty ("Vision API null response").
|
||||||
|
* Exported for unit tests.
|
||||||
*/
|
*/
|
||||||
export function extractChunkText(
|
export function extractChunkText(
|
||||||
chunk: LLMResponseChunk | null | undefined,
|
chunk: LLMResponseChunk | null | undefined,
|
||||||
): string {
|
): string {
|
||||||
if (!chunk) return "";
|
if (!chunk) return "";
|
||||||
const choice = chunk.choices?.[0];
|
const choice = chunk.choices?.[0];
|
||||||
|
const reasoningDetails = choice?.delta?.reasoning_details
|
||||||
|
?.map((d) => d.text ?? "")
|
||||||
|
.filter(Boolean)
|
||||||
|
.join("");
|
||||||
return (
|
return (
|
||||||
choice?.delta?.content ||
|
choice?.delta?.content ||
|
||||||
choice?.delta?.reasoning_content ||
|
choice?.delta?.reasoning_content ||
|
||||||
|
choice?.delta?.reasoning ||
|
||||||
|
reasoningDetails ||
|
||||||
choice?.message?.content ||
|
choice?.message?.content ||
|
||||||
choice?.text ||
|
choice?.text ||
|
||||||
chunk?.message?.content ||
|
chunk?.message?.content ||
|
||||||
|
|||||||
@@ -29,6 +29,35 @@ import {
|
|||||||
upsertCachedMediaByPhash,
|
upsertCachedMediaByPhash,
|
||||||
visionLruCache,
|
visionLruCache,
|
||||||
} from "./mediaCache.js";
|
} from "./mediaCache.js";
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Detect vision outputs where the model claims it saw no image at all
|
||||||
|
* ("Maaf, saya tidak melihat gambar apapun...", "Tidak ada gambar yang
|
||||||
|
* terlampir...", "I cannot see any image..."). Such text is NOT a valid
|
||||||
|
* analysis — caching it poisons the image cache for 24h (image/phash keys),
|
||||||
|
* so every re-analysis of the same image returns the "no image" text and the
|
||||||
|
* moderation LLM writes "lampiran gagal terbaca". These outputs must be
|
||||||
|
* treated as failures: never cached, and ignored when read back from cache.
|
||||||
|
*/
|
||||||
|
export function isNoImageSeenText(text: string | null | undefined): boolean {
|
||||||
|
if (!text) return false;
|
||||||
|
const lower = text.toLowerCase();
|
||||||
|
return (
|
||||||
|
/tidak (?:melihat|ada|terlihat) (?:gambar|foto|image)/i.test(lower) ||
|
||||||
|
/tidak (?:ada )?(?:gambar|foto|image) (?:apapun|yang terlampir)/i.test(
|
||||||
|
lower,
|
||||||
|
) ||
|
||||||
|
/gambar apapun/i.test(lower) ||
|
||||||
|
/tanpa (?:input )?(?:visual|gambar|image)/i.test(lower) ||
|
||||||
|
/\bno image (?:provided|attached|detected|found|was provided)?/i.test(
|
||||||
|
lower,
|
||||||
|
) ||
|
||||||
|
/(?:cannot|can't) see (?:any |an |the )?image/i.test(lower) ||
|
||||||
|
/i (?:do not|don't) (?:see|detect) (?:any |an |the )?image/i.test(lower) ||
|
||||||
|
/there (?:is|are) no image/i.test(lower)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
import {
|
import {
|
||||||
buildMediaCandidates,
|
buildMediaCandidates,
|
||||||
downloadAndExtractFrame,
|
downloadAndExtractFrame,
|
||||||
@@ -119,18 +148,32 @@ export const analyzeSingleMediaImage = async (
|
|||||||
|
|
||||||
// Layer 0: LRU
|
// Layer 0: LRU
|
||||||
const lruCached = visionLruCache.get(cacheKey);
|
const lruCached = visionLruCache.get(cacheKey);
|
||||||
if (lruCached) {
|
if (lruCached && !isNoImageSeenText(lruCached)) {
|
||||||
log.debug({ cacheKey }, "Vision LRU cache HIT (in-memory)");
|
log.debug({ cacheKey }, "Vision LRU cache HIT (in-memory)");
|
||||||
return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${lruCached}`;
|
return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${lruCached}`;
|
||||||
}
|
}
|
||||||
|
if (lruCached) {
|
||||||
|
// Poisoned entry ("I see no image") — drop it and re-analyze.
|
||||||
|
log.warn({ cacheKey }, "Vision LRU cache HIT was no-image-seen — dropping");
|
||||||
|
visionLruCache.delete(cacheKey);
|
||||||
|
}
|
||||||
|
|
||||||
// Layer 1: DB
|
// Layer 1: DB
|
||||||
const cached = await getCachedMediaAnalysis(cacheKey);
|
const cached = await getCachedMediaAnalysis(cacheKey);
|
||||||
if (cached) {
|
if (cached && !isNoImageSeenText(cached)) {
|
||||||
visionLruCache.set(cacheKey, cached);
|
visionLruCache.set(cacheKey, cached);
|
||||||
log.debug({ cacheKey }, "Media analysis cache HIT (DB → LRU)");
|
log.debug({ cacheKey }, "Media analysis cache HIT (DB → LRU)");
|
||||||
return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${cached}`;
|
return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${cached}`;
|
||||||
}
|
}
|
||||||
|
if (cached) {
|
||||||
|
// Poisoned DB entry — purge it so later messages re-analyze.
|
||||||
|
log.warn(
|
||||||
|
{ cacheKey },
|
||||||
|
"Media analysis cache HIT was no-image-seen — purging",
|
||||||
|
);
|
||||||
|
await deleteCachedMediaAnalysis(cacheKey).catch(() => {});
|
||||||
|
visionLruCache.delete(cacheKey);
|
||||||
|
}
|
||||||
|
|
||||||
// In-flight dedupe
|
// In-flight dedupe
|
||||||
const existing = inFlightVisionCalls.get(cacheKey);
|
const existing = inFlightVisionCalls.get(cacheKey);
|
||||||
@@ -173,7 +216,7 @@ export const analyzeSingleMediaImage = async (
|
|||||||
phash = await computeImagePhash(imgBuffer);
|
phash = await computeImagePhash(imgBuffer);
|
||||||
if (phash) {
|
if (phash) {
|
||||||
const phashCached = await getCachedMediaByPhash(phash);
|
const phashCached = await getCachedMediaByPhash(phash);
|
||||||
if (phashCached) {
|
if (phashCached && !isNoImageSeenText(phashCached)) {
|
||||||
visionLruCache.set(cacheKey, phashCached);
|
visionLruCache.set(cacheKey, phashCached);
|
||||||
await upsertCachedMediaAnalysis(
|
await upsertCachedMediaAnalysis(
|
||||||
cacheKey,
|
cacheKey,
|
||||||
@@ -183,6 +226,12 @@ export const analyzeSingleMediaImage = async (
|
|||||||
).catch(() => {});
|
).catch(() => {});
|
||||||
return phashCached;
|
return phashCached;
|
||||||
}
|
}
|
||||||
|
if (phashCached) {
|
||||||
|
log.warn(
|
||||||
|
{ phash, cacheKey },
|
||||||
|
"phash cache HIT was no-image-seen — ignoring",
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} catch {
|
} catch {
|
||||||
@@ -195,7 +244,7 @@ export const analyzeSingleMediaImage = async (
|
|||||||
for (let attempt = 0; attempt < 3; attempt++) {
|
for (let attempt = 0; attempt < 3; attempt++) {
|
||||||
try {
|
try {
|
||||||
const content = await llmVision(promptText, image.image_url);
|
const content = await llmVision(promptText, image.image_url);
|
||||||
if (content) {
|
if (content && !isNoImageSeenText(content)) {
|
||||||
await upsertCachedMediaAnalysis(
|
await upsertCachedMediaAnalysis(
|
||||||
cacheKey,
|
cacheKey,
|
||||||
content,
|
content,
|
||||||
@@ -213,7 +262,17 @@ export const analyzeSingleMediaImage = async (
|
|||||||
}
|
}
|
||||||
return content;
|
return content;
|
||||||
}
|
}
|
||||||
|
if (content) {
|
||||||
|
// Model claims it saw no image — same as a null response: NOT a
|
||||||
|
// valid analysis, and caching it would poison the key for every
|
||||||
|
// re-analysis of the same image (phash TTL is 7 days).
|
||||||
|
log.warn(
|
||||||
|
{ messageId, cacheKey },
|
||||||
|
"Vision returned no-image-seen text — not caching",
|
||||||
|
);
|
||||||
|
} else {
|
||||||
log.warn({ messageId }, "Vision API null response");
|
log.warn({ messageId }, "Vision API null response");
|
||||||
|
}
|
||||||
break;
|
break;
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
lastError = err instanceof Error ? err : new Error(String(err));
|
lastError = err instanceof Error ? err : new Error(String(err));
|
||||||
@@ -240,6 +299,7 @@ export const analyzeSingleMediaImage = async (
|
|||||||
"Vision failed after 3 attempts",
|
"Vision failed after 3 attempts",
|
||||||
);
|
);
|
||||||
await deleteCachedMediaAnalysis(cacheKey).catch(() => {});
|
await deleteCachedMediaAnalysis(cacheKey).catch(() => {});
|
||||||
|
visionLruCache.delete(cacheKey);
|
||||||
return FAILED_ANALYSIS_PREFIX;
|
return FAILED_ANALYSIS_PREFIX;
|
||||||
})();
|
})();
|
||||||
|
|
||||||
@@ -378,8 +438,7 @@ export async function prepareMediaMessage(
|
|||||||
// Profile is emitted ONCE per batch in a <user_profiles> map (see
|
// Profile is emitted ONCE per batch in a <user_profiles> map (see
|
||||||
// mediaBatchProcessor); here we only reference it to avoid repeating the
|
// mediaBatchProcessor); here we only reference it to avoid repeating the
|
||||||
// full summary on every message of the same user.
|
// full summary on every message of the same user.
|
||||||
const profileRef =
|
const profileRef = profile?.profile_summary?.trim()
|
||||||
profile && profile.profile_summary?.trim()
|
|
||||||
? buildUserProfileRef(target.user_id)
|
? buildUserProfileRef(target.user_id)
|
||||||
: "";
|
: "";
|
||||||
|
|
||||||
|
|||||||
@@ -33,6 +33,41 @@ describe("extractChunkText — streaming chunk text extraction", () => {
|
|||||||
).toBe("Task");
|
).toBe("Task");
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it('falls back to delta.reasoning — mimo via 9router streams reasoning there with content:""', () => {
|
||||||
|
// Exact shape seen from 9router → mimo-v2.5-free (2026-08-11):
|
||||||
|
// {"choices":[{"delta":{"content":"","reasoning":"The user wants a","role":"assistant"},"finish_reason":null,...}]}
|
||||||
|
expect(
|
||||||
|
extractChunkText({
|
||||||
|
choices: [
|
||||||
|
{
|
||||||
|
delta: { content: "", reasoning: "The user wants a" },
|
||||||
|
finish_reason: null,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}),
|
||||||
|
).toBe("The user wants a");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("joins delta.reasoning_details[].text when present", () => {
|
||||||
|
expect(
|
||||||
|
extractChunkText({
|
||||||
|
choices: [
|
||||||
|
{
|
||||||
|
delta: {
|
||||||
|
content: "",
|
||||||
|
reasoning: "",
|
||||||
|
reasoning_details: [
|
||||||
|
{ type: "reasoning.text", text: " detailed", index: 0 },
|
||||||
|
{ type: "reasoning.text", text: " description", index: 1 },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
finish_reason: null,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}),
|
||||||
|
).toBe(" detailed description");
|
||||||
|
});
|
||||||
|
|
||||||
it("prefers content over reasoning when both present (deepseek-style final answer)", () => {
|
it("prefers content over reasoning when both present (deepseek-style final answer)", () => {
|
||||||
expect(
|
expect(
|
||||||
extractChunkText({
|
extractChunkText({
|
||||||
|
|||||||
@@ -0,0 +1,52 @@
|
|||||||
|
// ═══════════════════════════════════════════════════════════════════════════
|
||||||
|
// isNoImageSeenText — vision outputs that claim "no image" must not be cached
|
||||||
|
// ═══════════════════════════════════════════════════════════════════════════
|
||||||
|
// Regression (2026-08-11): the vision model sometimes answered "Maaf, saya
|
||||||
|
// tidak melihat gambar apapun yang terlampir..." and that text was cached as
|
||||||
|
// a VALID vision_llm result. Every later analysis of the same image (same
|
||||||
|
// hash / phash) then hit the poisoned cache and the moderation LLM wrote
|
||||||
|
// "lampiran yang gagal terbaca" — image analysis seemed permanently broken
|
||||||
|
// even though 9router was responding fine.
|
||||||
|
import { describe, expect, it } from "vitest";
|
||||||
|
import { isNoImageSeenText } from "../src/modules/ai-moderation/visionAnalyzer.js";
|
||||||
|
|
||||||
|
describe("isNoImageSeenText — poisoned vision output detection", () => {
|
||||||
|
it("detects the exact poisoned strings seen in production", () => {
|
||||||
|
expect(
|
||||||
|
isNoImageSeenText(
|
||||||
|
"Maaf, saya tidak melihat gambar apapun yang terlampir dalam pesan Anda. Mohon kirimkan ulang gambarnya agar saya bisa mendeskripsikannya secara objektif dan spesifik.",
|
||||||
|
),
|
||||||
|
).toBe(true);
|
||||||
|
expect(
|
||||||
|
isNoImageSeenText(
|
||||||
|
"Tidak ada gambar yang terlampir. Tidak bisa deskripsi tanpa input visual.",
|
||||||
|
),
|
||||||
|
).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("detects English variants", () => {
|
||||||
|
expect(isNoImageSeenText("I cannot see any image in this message")).toBe(
|
||||||
|
true,
|
||||||
|
);
|
||||||
|
expect(isNoImageSeenText("No image provided")).toBe(true);
|
||||||
|
expect(isNoImageSeenText("there is no image attached")).toBe(true);
|
||||||
|
expect(isNoImageSeenText("I don't see an image")).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("does NOT flag legitimate image descriptions", () => {
|
||||||
|
expect(
|
||||||
|
isNoImageSeenText(
|
||||||
|
"Gambar ini menampilkan dua panel komik, seorang gadis berambut biru tersipu saat dipuji.",
|
||||||
|
),
|
||||||
|
).toBe(false);
|
||||||
|
expect(
|
||||||
|
isNoImageSeenText("Ini adalah screenshot dari sebuah website rekrutmen."),
|
||||||
|
).toBe(false);
|
||||||
|
expect(isNoImageSeenText("Emoji menampilkan ekspresi wajah tertawa.")).toBe(
|
||||||
|
false,
|
||||||
|
);
|
||||||
|
expect(isNoImageSeenText(null)).toBe(false);
|
||||||
|
expect(isNoImageSeenText(undefined)).toBe(false);
|
||||||
|
expect(isNoImageSeenText("")).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user