feat(moderation): fetch and analyze URLs (images and web text) from messages
- Added to safely extract and fetch up to 3 URLs per message (with SSRF protection, 5MB limit, and 8s timeout). - Implemented recursive extraction to resolve Tenor/Giphy links from their HTML viewers to raw GIF binaries. - In , fetched images are automatically injected as into the vision LLM context, and truncated webpage text is appended to the message string.
This commit is contained in:
@@ -9,6 +9,7 @@ import type {
|
|||||||
AttachmentRecord,
|
AttachmentRecord,
|
||||||
MessageRecord,
|
MessageRecord,
|
||||||
} from "./types.js";
|
} from "./types.js";
|
||||||
|
import { extractUrlsFromText, fetchUrlSafely } from "./urlFetcher.js";
|
||||||
|
|
||||||
const ModerationResponseSchema = z.object({
|
const ModerationResponseSchema = z.object({
|
||||||
results: z.array(
|
results: z.array(
|
||||||
@@ -468,6 +469,49 @@ export async function runModerationAnalysis(
|
|||||||
}),
|
}),
|
||||||
);
|
);
|
||||||
|
|
||||||
|
// --- Fetch URLs found in target messages ---
|
||||||
|
// To avoid slowing down the pipeline too much, we limit to 3 URLs per message.
|
||||||
|
const messageWebTextMap = new Map<string, string[]>();
|
||||||
|
|
||||||
|
await Promise.all(
|
||||||
|
targets.map(async (msg) => {
|
||||||
|
const content = msg.edited_content ?? msg.content;
|
||||||
|
const urls = extractUrlsFromText(content).slice(0, 3);
|
||||||
|
if (urls.length === 0) return;
|
||||||
|
|
||||||
|
const webTexts: string[] = [];
|
||||||
|
|
||||||
|
await Promise.all(
|
||||||
|
urls.map(async (url) => {
|
||||||
|
const result = await fetchUrlSafely(url);
|
||||||
|
|
||||||
|
if (result.type === "image" && result.data && result.mimeType) {
|
||||||
|
// Append as an image part
|
||||||
|
const dataUrl = `data:${result.mimeType};base64,${result.data.toString("base64")}`;
|
||||||
|
const part: RawImagePart = {
|
||||||
|
type: "image_url",
|
||||||
|
image_url: { url: dataUrl },
|
||||||
|
};
|
||||||
|
const existing = messageImageMap.get(msg.id) ?? [];
|
||||||
|
existing.push(part);
|
||||||
|
messageImageMap.set(msg.id, existing);
|
||||||
|
} else if (result.type === "text" && result.textContent) {
|
||||||
|
webTexts.push(`[Isi Web dari ${url}]: ${result.textContent}`);
|
||||||
|
} else if (result.type === "error") {
|
||||||
|
log.debug(
|
||||||
|
{ url, error: result.error },
|
||||||
|
"Failed to fetch URL for moderation context",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
|
||||||
|
if (webTexts.length > 0) {
|
||||||
|
messageWebTextMap.set(msg.id, webTexts);
|
||||||
|
}
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
|
||||||
const hasImages = messageImageMap.size > 0;
|
const hasImages = messageImageMap.size > 0;
|
||||||
|
|
||||||
// -------------------------------------------------------------------------
|
// -------------------------------------------------------------------------
|
||||||
@@ -571,7 +615,10 @@ CRITICAL: "message_id" HARUS berupa STRING (dibungkus tanda kutip ganda). Jangan
|
|||||||
const messagesBlock = targets
|
const messagesBlock = targets
|
||||||
.map((msg) => {
|
.map((msg) => {
|
||||||
const content = msg.edited_content ?? msg.content;
|
const content = msg.edited_content ?? msg.content;
|
||||||
return `[target] id=${msg.id} user=${msg.username}: ${content}`;
|
const webTexts = messageWebTextMap.get(msg.id) ?? [];
|
||||||
|
const webContext =
|
||||||
|
webTexts.length > 0 ? `\n${webTexts.join("\n")}` : "";
|
||||||
|
return `[target] id=${msg.id} user=${msg.username}: ${content}${webContext}`;
|
||||||
})
|
})
|
||||||
.join("\n");
|
.join("\n");
|
||||||
|
|
||||||
@@ -588,7 +635,10 @@ CRITICAL: "message_id" HARUS berupa STRING (dibungkus tanda kutip ganda). Jangan
|
|||||||
|
|
||||||
for (const msg of targets) {
|
for (const msg of targets) {
|
||||||
const content = msg.edited_content ?? msg.content;
|
const content = msg.edited_content ?? msg.content;
|
||||||
const msgText = `[target] id=${msg.id} user=${msg.username}: ${content}`;
|
const webTexts = messageWebTextMap.get(msg.id) ?? [];
|
||||||
|
const webContext = webTexts.length > 0 ? `\n${webTexts.join("\n")}` : "";
|
||||||
|
|
||||||
|
const msgText = `[target] id=${msg.id} user=${msg.username}: ${content}${webContext}`;
|
||||||
parts.push({ type: "text", text: msgText });
|
parts.push({ type: "text", text: msgText });
|
||||||
|
|
||||||
// Immediately follow the message text with its images
|
// Immediately follow the message text with its images
|
||||||
|
|||||||
@@ -0,0 +1,207 @@
|
|||||||
|
import { resolve } from "node:dns/promises";
|
||||||
|
import { isIP } from "node:net";
|
||||||
|
import { createChildLogger } from "../logger.js";
|
||||||
|
|
||||||
|
const log = createChildLogger("urlFetcher");
|
||||||
|
|
||||||
|
export interface FetchedUrlContext {
|
||||||
|
url: string;
|
||||||
|
type: "image" | "text" | "error";
|
||||||
|
data?: Buffer;
|
||||||
|
mimeType?: string;
|
||||||
|
textContent?: string;
|
||||||
|
error?: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
const MAX_FETCH_SIZE = 5 * 1024 * 1024; // 5 MB
|
||||||
|
const FETCH_TIMEOUT_MS = 8000;
|
||||||
|
const URL_REGEX = /https?:\/\/[^\s<]+[^<.,:;"')\]\s]/gi;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Basic SSRF protection.
|
||||||
|
* Note: A sophisticated attacker could still use DNS rebinding.
|
||||||
|
*/
|
||||||
|
async function isSafeUrl(urlStr: string): Promise<boolean> {
|
||||||
|
try {
|
||||||
|
const parsed = new URL(urlStr);
|
||||||
|
const host = parsed.hostname;
|
||||||
|
|
||||||
|
// Block obvious local IPs/hostnames
|
||||||
|
if (
|
||||||
|
host === "localhost" ||
|
||||||
|
host === "127.0.0.1" ||
|
||||||
|
host === "::1" ||
|
||||||
|
host.startsWith("192.168.") ||
|
||||||
|
host.startsWith("10.") ||
|
||||||
|
/^172\.(1[6-9]|2[0-9]|3[0-1])\./.test(host)
|
||||||
|
) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Try resolving to check if it resolves to a local IP
|
||||||
|
if (!isIP(host)) {
|
||||||
|
try {
|
||||||
|
const addresses = await resolve(host);
|
||||||
|
for (const ip of addresses) {
|
||||||
|
if (
|
||||||
|
ip === "127.0.0.1" ||
|
||||||
|
ip.startsWith("192.168.") ||
|
||||||
|
ip.startsWith("10.") ||
|
||||||
|
/^172\.(1[6-9]|2[0-9]|3[0-1])\./.test(ip)
|
||||||
|
) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} catch (err) {
|
||||||
|
// If DNS fails, we can't fetch it anyway
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return true;
|
||||||
|
} catch (err) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function extractOgImage(html: string): string | null {
|
||||||
|
// Look for <meta ... property="og:image" ... content="..."> or <meta ... name="twitter:image" ... content="...">
|
||||||
|
const ogRegex = /<meta[^>]*(?:property|name)=["'](?:og:image|twitter:image)["'][^>]*content=["']([^"']+)["']/i;
|
||||||
|
const match = html.match(ogRegex);
|
||||||
|
if (match && match[1]) {
|
||||||
|
// Unescape basic HTML entities
|
||||||
|
return match[1].replace(/&/g, "&").replace(/"/g, '"');
|
||||||
|
}
|
||||||
|
|
||||||
|
// Try reversed attribute order: <meta ... content="..." ... property="og:image">
|
||||||
|
const ogRegexRev = /<meta[^>]*content=["']([^"']+)["'][^>]*(?:property|name)=["'](?:og:image|twitter:image)["']/i;
|
||||||
|
const matchRev = html.match(ogRegexRev);
|
||||||
|
if (matchRev && matchRev[1]) {
|
||||||
|
return matchRev[1].replace(/&/g, "&").replace(/"/g, '"');
|
||||||
|
}
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
function truncateAndCleanHtml(html: string, maxLen = 1000): string {
|
||||||
|
// Strip <script> and <style> entirely
|
||||||
|
let text = html.replace(
|
||||||
|
/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/gi,
|
||||||
|
" ",
|
||||||
|
);
|
||||||
|
text = text.replace(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/gi, " ");
|
||||||
|
// Strip all other HTML tags
|
||||||
|
text = text.replace(/<[^>]+>/g, " ");
|
||||||
|
// Replace multiple spaces/newlines
|
||||||
|
text = text.replace(/\s+/g, " ").trim();
|
||||||
|
|
||||||
|
return text.substring(0, maxLen);
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function fetchUrlSafely(
|
||||||
|
url: string,
|
||||||
|
depth = 0,
|
||||||
|
): Promise<FetchedUrlContext> {
|
||||||
|
if (depth > 1) {
|
||||||
|
return { url, type: "error", error: "Max redirect/meta depth reached" };
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!(await isSafeUrl(url))) {
|
||||||
|
return { url, type: "error", error: "Unsafe URL blocked" };
|
||||||
|
}
|
||||||
|
|
||||||
|
const controller = new AbortController();
|
||||||
|
const timeoutId = setTimeout(() => controller.abort(), FETCH_TIMEOUT_MS);
|
||||||
|
|
||||||
|
try {
|
||||||
|
const response = await fetch(url, {
|
||||||
|
signal: controller.signal,
|
||||||
|
headers: {
|
||||||
|
"User-Agent":
|
||||||
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36 DiscordBot/2.0",
|
||||||
|
Accept: "image/webp,image/apng,image/*,*/*;q=0.8",
|
||||||
|
},
|
||||||
|
// Do not follow more than a few redirects natively, fetch handles up to 20 by default
|
||||||
|
});
|
||||||
|
|
||||||
|
if (!response.ok) {
|
||||||
|
return { url, type: "error", error: `HTTP ${response.status}` };
|
||||||
|
}
|
||||||
|
|
||||||
|
const contentType = response.headers.get("content-type") || "";
|
||||||
|
const contentLength = parseInt(
|
||||||
|
response.headers.get("content-length") || "0",
|
||||||
|
10,
|
||||||
|
);
|
||||||
|
|
||||||
|
if (contentLength > MAX_FETCH_SIZE) {
|
||||||
|
return { url, type: "error", error: "Content too large" };
|
||||||
|
}
|
||||||
|
|
||||||
|
const buffer = await response.arrayBuffer();
|
||||||
|
if (buffer.byteLength > MAX_FETCH_SIZE) {
|
||||||
|
return { url, type: "error", error: "Downloaded content too large" };
|
||||||
|
}
|
||||||
|
|
||||||
|
if (contentType.startsWith("image/")) {
|
||||||
|
return {
|
||||||
|
url,
|
||||||
|
type: "image",
|
||||||
|
data: Buffer.from(buffer),
|
||||||
|
mimeType: contentType,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
if (
|
||||||
|
contentType.startsWith("text/html") ||
|
||||||
|
contentType.startsWith("text/plain")
|
||||||
|
) {
|
||||||
|
const text = Buffer.from(buffer).toString("utf-8");
|
||||||
|
|
||||||
|
// If it's HTML, try to find an og:image first (for Tenor/Giphy etc)
|
||||||
|
if (contentType.startsWith("text/html")) {
|
||||||
|
const ogImage = extractOgImage(text);
|
||||||
|
if (ogImage && ogImage.startsWith("http")) {
|
||||||
|
// Fetch the og:image instead
|
||||||
|
return fetchUrlSafely(ogImage, depth + 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Fallback to text content
|
||||||
|
const cleaned = truncateAndCleanHtml(text, 1000);
|
||||||
|
return {
|
||||||
|
url,
|
||||||
|
type: "text",
|
||||||
|
textContent: cleaned,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
url,
|
||||||
|
type: "error",
|
||||||
|
error: `Unsupported content type: ${contentType}`,
|
||||||
|
};
|
||||||
|
} catch (err) {
|
||||||
|
return {
|
||||||
|
url,
|
||||||
|
type: "error",
|
||||||
|
error: err instanceof Error ? err.message : String(err),
|
||||||
|
};
|
||||||
|
} finally {
|
||||||
|
clearTimeout(timeoutId);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export function extractUrlsFromText(text: string): string[] {
|
||||||
|
const matches = text.match(URL_REGEX);
|
||||||
|
if (!matches) return [];
|
||||||
|
// Deduplicate and filter out things that obviously aren't valid
|
||||||
|
return Array.from(new Set(matches)).filter((url) => {
|
||||||
|
try {
|
||||||
|
new URL(url);
|
||||||
|
return true;
|
||||||
|
} catch {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user