fix(gateway): image vision analysis + media cache lock failures
Two root causes behind 'all image analysis failing':
1. imageResizer still emitted lossless PNG for vision input. A 1024px
Facebook photo balloons to multi-MB PNG base64 that the vision model
silently rejects ('Vision API null response'). Switch to JPEG q85
(no upscaling) — same photo drops to ~100-400KB, model processes fine.
Re-encodes even already-small images so raw originals never bloat the
data URL. Added tests/imageResizer.test.ts covering both cases.
2. acquireMediaAnalysisLock INSERT aborted with 'index row requires N
bytes, maximum size is 8191'. text_analysis_cache.text is the PK in a
B-tree index (8191-byte/row cap); callers pass the raw image URL as the
key, and base64 data URLs / very long URLs blow past the limit, so the
lock INSERT fails and every media analysis is skipped. Hash the URL in
makeImageCacheKey (image:<sha256[:32]>) — fixed-length, deterministic,
well under the limit. All store/get/lock/delete callers already route
through this function so lookup stays consistent.
This commit is contained in:
@@ -77,14 +77,16 @@ export function makeCustomEmojiCacheKey(emojiId: string): string {
|
||||
* (different URLs) never collide.
|
||||
*/
|
||||
export function makeImageCacheKey(imageUrl: string): string {
|
||||
try {
|
||||
const u = new URL(imageUrl);
|
||||
u.search = "";
|
||||
u.hash = "";
|
||||
return `image:${u.toString()}`;
|
||||
} catch {
|
||||
return `image:${imageUrl}`;
|
||||
}
|
||||
// Hash the URL to a fixed-length key. The raw Discord CDN URL is short,
|
||||
// but callers sometimes pass base64 data URLs (can be multi-MB) or very
|
||||
// long signed/external URLs. text_analysis_cache.text is the PK and lives
|
||||
// in a B-tree index with an 8191-byte per-row limit — inserting a long URL
|
||||
// as the key aborts the whole INSERT ("index row requires N bytes, maximum
|
||||
// size is 8191"), which fails acquireMediaAnalysisLock and silently skips
|
||||
// every media analysis. A 32-char sha256 keeps the key well under the limit
|
||||
// and is still deterministic (same attachment → same key).
|
||||
const hash = createHash("sha256").update(imageUrl).digest("hex").slice(0, 32);
|
||||
return `image:${hash}`;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -4,10 +4,17 @@ import { createChildLogger } from "@/shared/logger/index";
|
||||
const log = createChildLogger("imageResizer");
|
||||
|
||||
/**
|
||||
* Prepare an image buffer for optimal vision LLM analysis.
|
||||
* Prepare an image buffer for vision LLM analysis.
|
||||
*
|
||||
* - Resizes to maxDim x maxDim maintaining aspect ratio (only if larger)
|
||||
* - Converts to PNG (lossless) to preserve full image detail
|
||||
* - Resizes to maxDim x maxDim maintaining aspect ratio WITHOUT upscaling
|
||||
* (small images such as stickers/emojis are passed through at original size,
|
||||
* just re-encoded)
|
||||
* - Encodes as JPEG (lossy, quality ~85). Photos compress VERY poorly in
|
||||
* lossless PNG — a 1024px Facebook photo balloons to multi-MB PNG base64 that
|
||||
* the vision model silently rejects (empty response → "Vision API null").
|
||||
* JPEG keeps the same photo at ~100–400KB, which the model processes fine and
|
||||
* stays well below request/token size limits.
|
||||
|
||||
* - Falls back to original buffer if sharp fails
|
||||
*
|
||||
* @param buf - Raw image buffer
|
||||
@@ -20,35 +27,28 @@ export async function resizeImageForVision(
|
||||
): Promise<{ data: Buffer; mimeType: string }> {
|
||||
try {
|
||||
const metadata = await sharp(buf).metadata();
|
||||
const inputFormat = metadata.format ?? "jpeg";
|
||||
|
||||
// Skip resize entirely if already within max dimension
|
||||
if ((metadata.width ?? 0) <= maxDim && (metadata.height ?? 0) <= maxDim) {
|
||||
return { data: buf, mimeType: `image/${inputFormat}` };
|
||||
}
|
||||
|
||||
// Resize dimension only — convert to PNG lossless to preserve detail
|
||||
// Always (re-)encode to JPEG and fit inside maxDim without upscaling.
|
||||
// Skipping the encode for already-small images left raw originals in
|
||||
// their native (often lossless PNG or full-quality) form, which could
|
||||
// still bloat data URLs and trip the vision model's size limit.
|
||||
const resized = await sharp(buf)
|
||||
.resize(maxDim, maxDim, {
|
||||
fit: "inside",
|
||||
withoutEnlargement: true,
|
||||
})
|
||||
.png()
|
||||
.resize(maxDim, maxDim, { fit: "inside", withoutEnlargement: true })
|
||||
.jpeg({ quality: 85 })
|
||||
.toBuffer();
|
||||
|
||||
const inputFormat = metadata.format ?? "jpeg";
|
||||
log.debug(
|
||||
{
|
||||
originalSize: buf.length,
|
||||
originalFormat: inputFormat,
|
||||
resizedSize: resized.length,
|
||||
reductionPct: Math.round(
|
||||
((buf.length - resized.length) / buf.length) * 100,
|
||||
),
|
||||
reductionPct: Math.round(((buf.length - resized.length) / buf.length) * 100),
|
||||
},
|
||||
"Image resized for vision analysis (lossless PNG)",
|
||||
"Image resized for vision analysis (JPEG)",
|
||||
);
|
||||
|
||||
return { data: resized, mimeType: "image/png" };
|
||||
return { data: resized, mimeType: "image/jpeg" };
|
||||
} catch (error) {
|
||||
log.warn(
|
||||
{ error: error instanceof Error ? error.message : String(error) },
|
||||
|
||||
Reference in New Issue
Block a user