/** * visionAnalyzer.ts * * Vision analysis for media content — prepares media messages for the * moderation pipeline, runs single-image vision LLM analysis with * multi-layer caching, and detects whether a message has media content. */ import { createChildLogger } from "@/shared/logger/index"; import { delay } from "@/shared/utils/index"; import { config } from "../../shared/config/config.js"; import { extractMessageMediaEvidence } from "../message-capture/messageMetadata.js"; import type { AttachmentRecord, MessageRecord, } from "../message-capture/types.js"; import { llmVision } from "./llmClient.js"; import { acquireMediaAnalysisLock, deleteCachedMediaAnalysis, FAILED_ANALYSIS_PREFIX, getCachedMediaAnalysis, inFlightVisionCalls, makeCustomEmojiCacheKey, makeImageCacheKey, makeStickerCacheKey, upsertCachedMediaAnalysis, visionLruCache, } from "./mediaCache.js"; /** * Detect vision outputs where the model claims it saw no image at all * ("Maaf, saya tidak melihat gambar apapun...", "Tidak ada gambar yang * terlampir...", "I cannot see any image..."). Such text is NOT a valid * analysis — caching it poisons the image cache for 24h (image/phash keys), * so every re-analysis of the same image returns the "no image" text and the * moderation LLM writes "lampiran gagal terbaca". These outputs must be * treated as failures: never cached, and ignored when read back from cache. */ export function isNoImageSeenText(text: string | null | undefined): boolean { if (!text) return false; const lower = text.toLowerCase(); return ( /tidak (?:melihat|ada|terlihat) (?:gambar|foto|image)/i.test(lower) || /tidak (?:ada )?(?:gambar|foto|image) (?:apapun|yang terlampir)/i.test( lower, ) || /gambar apapun/i.test(lower) || /tanpa (?:input )?(?:visual|gambar|image)/i.test(lower) || /\bno image (?:provided|attached|detected|found|was provided)?/i.test( lower, ) || /(?:cannot|can't) see (?:any |an |the )?image/i.test(lower) || /i (?:do not|don't) (?:see|detect) (?:any |an |the )?image/i.test(lower) || /there (?:is|are) no image/i.test(lower) ); } import { buildMediaCandidates, downloadAndExtractFrame, downloadMediaCandidate, fetchUrlInline, } from "./mediaDownloader.js"; import { buildReferenceXml, buildUserProfileRef, escapeXml, formatReputationAttrs, getAnalysisContent, resolveDisplayName, resolveIsBot, resolveIsEdited, truncateForAi, } from "./moderationBuilders.js"; import { buildCustomEmojiVisionPrompt, buildGeneralImageVisionPrompt, buildStickerTextOnlyWarning, buildStickerVisionPrompt, } from "./moderationPrompt.js"; import { extractSearchQueries, formatSearchResults, searchSearxng, } from "./searxngSearch.js"; import { buildTermGlossaryBlock } from "./termGlossary.js"; import { extractUrlsFromText } from "./urlFetcher.js"; import { getUserProfile } from "./userProfileStore.js"; import { initializeUserReputation } from "./userReputationStore.js"; // --------------------------------------------------------------------------- // Types // --------------------------------------------------------------------------- export type MessageImagePart = { type: "image_url"; image_url: { url: string }; sourceLabel: string; stickerName?: string; customEmojiId?: string; customEmojiName?: string; }; export interface PreparedMediaMessage { targetId: string; messageBlock: string; } // --------------------------------------------------------------------------- // Media detection // --------------------------------------------------------------------------- export function hasMediaContent( target: MessageRecord, attachments?: AttachmentRecord[], ): boolean { if (target.metadata) { const evidence = extractMessageMediaEvidence(target.metadata); if ( evidence.stickers.length > 0 || evidence.embeds.length > 0 || evidence.attachments.length > 0 ) return true; } if (attachments?.some((a) => a.message_id === target.id)) return true; return false; } // --------------------------------------------------------------------------- // Single-image vision analysis // --------------------------------------------------------------------------- export const analyzeSingleMediaImage = async ( messageId: string, image: MessageImagePart, ): Promise => { const cacheKey = image.customEmojiId ? makeCustomEmojiCacheKey(image.customEmojiId) : image.stickerName ? makeStickerCacheKey(image.stickerName) : makeImageCacheKey(image.image_url.url); const log = createChildLogger("mediaAnalysis"); // Layer 0: LRU const lruCached = visionLruCache.get(cacheKey); if (lruCached && !isNoImageSeenText(lruCached)) { log.debug({ cacheKey }, "Vision LRU cache HIT (in-memory)"); return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${lruCached}`; } if (lruCached) { // Poisoned entry ("I see no image") — drop it and re-analyze. log.warn({ cacheKey }, "Vision LRU cache HIT was no-image-seen — dropping"); visionLruCache.delete(cacheKey); } // Layer 1: DB const cached = await getCachedMediaAnalysis(cacheKey); if (cached && !isNoImageSeenText(cached)) { visionLruCache.set(cacheKey, cached); log.debug( { cacheKey, messageId, cachedLen: cached.length }, "Media analysis cache HIT (DB → LRU)", ); return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${cached}`; } if (cached) { // Poisoned DB entry — purge it so later messages re-analyze. log.warn( { cacheKey }, "Media analysis cache HIT was no-image-seen — purging", ); await deleteCachedMediaAnalysis(cacheKey).catch(() => {}); visionLruCache.delete(cacheKey); } // In-flight dedupe const existing = inFlightVisionCalls.get(cacheKey); if (existing) { log.debug({ cacheKey }, "Media analysis in-flight dedupe"); const result = await existing; return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${result}`; } const promptText = image.stickerName ? buildStickerVisionPrompt(image.stickerName, messageId) : image.customEmojiName ? buildCustomEmojiVisionPrompt(image.customEmojiName, messageId) : buildGeneralImageVisionPrompt(image.sourceLabel, messageId); const visionPromise = (async (): Promise => { // Distributed lock const locked = await acquireMediaAnalysisLock(cacheKey, Date.now() + 60000); if (!locked) { log.debug({ cacheKey }, "Distributed lock — polling"); for (let i = 0; i < 15; i++) { await new Promise((r) => setTimeout(r, 2000)); const polled = await getCachedMediaAnalysis(cacheKey); if (polled) { visionLruCache.set(cacheKey, polled); return polled; } } log.warn({ cacheKey }, "Distributed lock polling timed out"); return FAILED_ANALYSIS_PREFIX; } // Vision API call let lastError: Error | null = null; for (let attempt = 0; attempt < 3; attempt++) { try { const content = await llmVision(promptText, image.image_url); if (content && !isNoImageSeenText(content)) { // Defensive: log when a vision analysis is cached so we can trace // if the SAME analysis text is being stored for DIFFERENT cache keys // (which would indicate the vision model is returning duplicates). log.debug( { cacheKey, messageId, contentLen: content.length }, "Vision analysis cached (new entry)", ); await upsertCachedMediaAnalysis( cacheKey, content, "vision_llm", Date.now() + 24 * 60 * 60 * 1000, ); visionLruCache.set(cacheKey, content); return content; } if (content) { // Model claims it saw no image — same as a null response: NOT a // valid analysis, and caching it would poison the cache key. log.warn( { messageId, cacheKey }, "Vision returned no-image-seen text — not caching", ); } else { log.warn({ messageId }, "Vision API null response"); } break; } catch (err) { lastError = err instanceof Error ? err : new Error(String(err)); if (attempt < 2) { const backoffMs = Math.min( 2_000 * 3 ** attempt + Math.random() * 500, 30_000, ); log.warn( { messageId, attempt: attempt + 1, backoffMs, error: lastError.message, }, "Vision retry", ); await delay(backoffMs); } } } log.warn( { messageId, lastError: lastError?.message ?? "null" }, "Vision failed after 3 attempts", ); await deleteCachedMediaAnalysis(cacheKey).catch(() => {}); visionLruCache.delete(cacheKey); return FAILED_ANALYSIS_PREFIX; })(); inFlightVisionCalls.set(cacheKey, visionPromise); try { const content = await visionPromise; return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${content}`; } catch (outerErr) { log.error( { messageId, cacheKey, error: outerErr instanceof Error ? outerErr.message : String(outerErr), }, "visionPromise threw unexpectedly", ); return `[Media analysis for message ${messageId}] ${image.sourceLabel}: ${FAILED_ANALYSIS_PREFIX}`; } finally { inFlightVisionCalls.delete(cacheKey); } }; // --------------------------------------------------------------------------- // Media message preparation // --------------------------------------------------------------------------- /** * Download images, run vision analysis, and build the message XML block * for a single media-bearing message. Does NOT make the moderation LLM call. */ export async function prepareMediaMessage( target: MessageRecord, allAttachments: AttachmentRecord[] | undefined, ): Promise { const _log = createChildLogger("mediaAnalysis"); const targetId = target.id; const imageMap = new Map(); const webTextMap = new Map(); const mediaAnalysisMap = new Map(); const maxDimension = config.AI_LLM_IMAGE_MAX_DIMENSION ?? 1024; const content = getAnalysisContent(target); const downloadPromises: Array> = []; // Attachments const msgAttachments = (allAttachments ?? []) .filter( (a) => a.message_id === targetId && (a.uploaded_url ?? a.discord_url ?? null) && (a.type.startsWith("image/") || a.type.startsWith("video/")), ) .slice(0, 8); for (const att of msgAttachments) { downloadPromises.push( downloadAndExtractFrame(att, targetId, maxDimension, imageMap), ); } // URLs const urls = extractUrlsFromText(content).slice(0, 3); const urlWebTexts: string[] = []; for (const url of urls) { downloadPromises.push( fetchUrlInline(url, targetId, maxDimension, imageMap, urlWebTexts), ); } // Stickers, embeds, custom emoji const mediaEvidence = extractMessageMediaEvidence(target.metadata); for (const candidate of buildMediaCandidates(targetId, mediaEvidence)) { downloadPromises.push( downloadMediaCandidate( candidate, targetId, maxDimension, imageMap, mediaAnalysisMap, ), ); } await Promise.all(downloadPromises); if (urlWebTexts.length > 0) webTextMap.set(targetId, urlWebTexts); // Vision analysis await Promise.all( Array.from(imageMap.entries()).flatMap(([msgId, images]) => images.map(async (image) => { const summary = await analyzeSingleMediaImage(msgId, image); const existing = mediaAnalysisMap.get(msgId) ?? []; existing.push(summary); mediaAnalysisMap.set(msgId, existing); }), ), ); // SearXNG let searxngXml = ""; const queries = extractSearchQueries(content); if (queries.length > 0) { const results = await Promise.allSettled( queries.map((q) => searchSearxng(q)), ); const parts: string[] = []; for (let i = 0; i < results.length; i++) { const r = results[i]; if (r.status === "fulfilled" && r.value.length > 0) parts.push(formatSearchResults(r.value)); } if (parts.length > 0) searxngXml = `\n\n${parts.join("\n")}\n`; } // Term glossary — cached per-word Wikipedia definitions for words the LLM // may not know. Bounded and cached (in-memory + Redis), so this adds no // meaningful latency to the media path either. const glossaryXml = await buildTermGlossaryBlock([content]).catch(() => ""); const glossaryCtx = glossaryXml ? `\n${glossaryXml}` : ""; // Build XML block const webTexts = webTextMap.get(targetId) ?? []; const mediaAnalyses = mediaAnalysisMap.get(targetId) ?? []; const webContext = webTexts.length > 0 ? `\n${webTexts.join("\n")}` : ""; const mediaAnalysisContext = mediaAnalyses.length > 0 ? `\n${mediaAnalyses.join("\n")}` : ""; const mediaContext = [ mediaEvidence.stickers.length > 0 ? mediaEvidence.stickers .map((s) => buildStickerTextOnlyWarning(s.name, s.url)) .join(" ") : null, mediaEvidence.embeds.length > 0 ? `[embed evidence: ${mediaEvidence.embeds.map((e) => [e.title, e.description, e.url, e.image, e.thumbnail].filter(Boolean).join(" | ")).join(" || ")}]` : null, ] .filter(Boolean) .join(" "); const rep = await initializeUserReputation(target.user_id, target.guild_id); const profile = await getUserProfile(target.user_id); const refXml = await buildReferenceXml(target); // Profile is emitted ONCE per batch in a map (see // mediaBatchProcessor); here we only reference it to avoid repeating the // full summary on every message of the same user. const profileRef = profile?.profile_summary?.trim() ? buildUserProfileRef(target.user_id) : ""; // Rich reputation — attrs only, no user history injection (per channel context preference) const repAttrs = formatReputationAttrs(rep); const repXml = ``; const isBot = resolveIsBot(target); const isEdited = resolveIsEdited(target); const messageBlock = `\n ${repXml}${profileRef ? `\n ${profileRef}` : ""}${refXml ? `\n ${refXml}` : ""}\n ${escapeXml(truncateForAi(content))}${mediaContext ? ` ${escapeXml(mediaContext)}` : ""}${webContext}${mediaAnalysisContext}${searxngXml}${glossaryCtx}\n`; return { targetId, messageBlock }; }