fix(moderation): immediately abort retries on 429 Too Many Requests
- In llmModerationClient.ts (inner retry), if OpenAI throws a 429 (or 401/403), throw p-retry's AbortError to immediately exit the 3-attempt inner retry loop. - In aiAnalyzer.ts (outer retry), propagate the AbortError from runModerationAnalysis so the 2-attempt outer retry loop also aborts immediately. - This ensures that a burst of 20 concurrent tasks hitting rate limits immediately returns the messages to the DB queue (as 'analysis_incomplete') and rapidly increments the individual circuit breaker, pausing processing and preventing a thundering herd instead of making 12 API calls per stuck message.
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
import { existsSync } from "node:fs";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { AbortError } from "p-retry";
|
||||
import { Piscina } from "piscina";
|
||||
import { config } from "../config.js";
|
||||
import { createChildLogger } from "../logger.js";
|
||||
@@ -227,6 +228,7 @@ async function processIndividualFallback(
|
||||
|
||||
const analysisResult = await retryWithBackoff(
|
||||
async () => {
|
||||
try {
|
||||
const result = await runModerationAnalysis({
|
||||
targets: [message],
|
||||
contextText: contextLines.join("\n"),
|
||||
@@ -248,7 +250,22 @@ async function processIndividualFallback(
|
||||
|
||||
// Got a real result — clear the incomplete flag.
|
||||
exhaustedOnIncomplete = false;
|
||||
|
||||
return result;
|
||||
} catch (err: any) {
|
||||
// Propagate AbortError so outer retry is immediately cancelled on 429.
|
||||
if (err instanceof AbortError) {
|
||||
throw err;
|
||||
}
|
||||
if (
|
||||
err?.status === 429 ||
|
||||
err?.status === 401 ||
|
||||
err?.status === 403
|
||||
) {
|
||||
throw new AbortError(err);
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
},
|
||||
{
|
||||
retries: 2,
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import OpenAI from "openai";
|
||||
import { AbortError } from "p-retry";
|
||||
import { z } from "zod";
|
||||
import { config } from "../config.js";
|
||||
import { createChildLogger } from "../logger.js";
|
||||
@@ -34,7 +35,10 @@ const openai = new OpenAI({
|
||||
|
||||
// Override headers to bypass Cloudflare WAF Bot Fight Mode
|
||||
const headers = new Headers(init?.headers);
|
||||
headers.set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36");
|
||||
headers.set(
|
||||
"User-Agent",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||
);
|
||||
for (const key of Array.from(headers.keys())) {
|
||||
if (key.toLowerCase().startsWith("x-stainless")) {
|
||||
headers.delete(key);
|
||||
@@ -610,6 +614,7 @@ CRITICAL: "message_id" HARUS berupa STRING (dibungkus tanda kutip ganda). Jangan
|
||||
try {
|
||||
const analysis = await retryWithBackoff(
|
||||
async () => {
|
||||
try {
|
||||
const completion = await openai.chat.completions.create({
|
||||
model: config.AI_LLM_MODEL,
|
||||
messages: [
|
||||
@@ -666,6 +671,18 @@ CRITICAL: "message_id" HARUS berupa STRING (dibungkus tanda kutip ganda). Jangan
|
||||
);
|
||||
throw parseError;
|
||||
}
|
||||
} catch (apiError: any) {
|
||||
// Immediately abort retries on rate limits or auth errors so the
|
||||
// message can return to the DB queue instead of bursting retries.
|
||||
if (
|
||||
apiError?.status === 429 ||
|
||||
apiError?.status === 401 ||
|
||||
apiError?.status === 403
|
||||
) {
|
||||
throw new AbortError(apiError);
|
||||
}
|
||||
throw apiError;
|
||||
}
|
||||
},
|
||||
{
|
||||
retries: 3,
|
||||
|
||||
Reference in New Issue
Block a user