feat(gateway): refactor term extraction and scoring logic into textSignals.ts for reuse

This commit is contained in:
asepharyana
2026-08-31 22:59:27 +07:00
parent 12cc956329
commit 0e31aa06b8
4 changed files with 288 additions and 149 deletions
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -22,6 +22,12 @@ import { createChildLogger } from "@/shared/logger/index";
import { createAbortControllerWithTimeout } from "@/shared/utils/index"; import { createAbortControllerWithTimeout } from "@/shared/utils/index";
import { config } from "../../shared/config/config.js"; import { config } from "../../shared/config/config.js";
import { cacheGet, cacheSet, makeCacheKey } from "./cacheStore.js"; import { cacheGet, cacheSet, makeCacheKey } from "./cacheStore.js";
import {
cleanContent,
isKnownTerm,
isMostlyStopwords,
scoreWord,
} from "./textSignals.js";
const log = createChildLogger("wikipedia-client"); const log = createChildLogger("wikipedia-client");
@@ -206,73 +212,121 @@ export async function wikipediaSummary(
} }
} }
/** export interface ExtractSearchQueryOptions {
* Extract meaningful search queries from message content. maxQueries?: number;
* Uses multiple strategies to find terms worth searching. }
* Returns up to 3 clean queries.
*/
export function extractSearchQueries(content: string): string[] {
const queries = new Set<string>();
// 1. Quoted phrases (explicit user intent) /**
const quotedPhrases = content.match(/"([^"]+)"|'([^']+)'/g); * Verbs that signal "find/watch/download X" — used only to find a phrase
* BOUNDARY. Unlike the old version, no trailing category word
* ("anime|film|series") is required, so coverage isn't capped by an
* enumerated category list. The capture stops at the first coordinating
* conjunction (or end of message) so "nonton X sama Y terus Z" yields the
* single entity X instead of swallowing the whole multi-entity tail into
* one unsearchable blob.
*/
const SEARCH_INTENT_VERBS =
/\b(?:nonton|tonton|rekomen(?:dasiin|dasikan)?|cari(?:in|kan)?|search|google|download|donlod|unduh|streaming|baca|dengerin|dengar(?:kan)?)\b\s+(.+?)(?=\s+(?:sama|dan|juga|terus|lalu|atau|and|or)\b|\s*[!?.]*$)/i;
/** "apa itu X" / "arti X" / "what is X" — factual/definition intent. */
const DEFINITION_INTENT =
/\b(?:apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti(?:nya)?|meaning(?:\s+of)?|definisi(?:nya)?|definition(?:\s+of)?)\s+(.{3,80})/i;
/** Multi-word capitalized runs — usually a title/named entity regardless of
* surrounding verbs (e.g. "Attack on Titan", "One Piece"). No trigger word
* needed at all. */
const PROPER_NOUN_PHRASE = /\b([A-Z][\p{L}]*(?:\s+[A-Z][\p{L}]*){1,4})\b/gu;
interface PhraseCandidate {
phrase: string;
score: number;
}
/**
* Scores a phrase with the SAME per-word signals as the term glossary
* (proper-noun casing, foreign spelling, hyphenation — see textSignals.ts),
* plus a bonus for the extraction strategy that surfaced it. Returns null
* for junk (mostly stopwords, or made entirely of known-safe terms like
* "Discord"/"Google" — those never need a Wikipedia lookup).
*/
function scorePhrase(rawPhrase: string, bonus: number): PhraseCandidate | null {
const clean = rawPhrase
.trim()
.replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "");
if (clean.length < 2 || clean.length > 80) return null;
if (isMostlyStopwords(clean)) return null;
const words = clean.split(/\s+/);
let score = bonus;
let hasNonStopword = false;
for (const w of words) {
if (isKnownTerm(w.toLowerCase())) continue;
hasNonStopword = true;
score += scoreWord(w);
}
if (!hasNonStopword) return null;
return { phrase: clean, score };
}
/**
* Extracts candidate phrases worth searching on Wikipedia — scored by the
* same word-level signals as the term glossary, instead of requiring an
* exact match against a fixed, manually-maintained category/brand list.
* Pure CPU-side regex + scoring — no network or LLM call, so using this
* more broadly never adds AI requests.
*
* Returns up to `maxQueries` phrases (default 3), highest-scored first.
*/
export function extractSearchQueries(
content: string,
options: ExtractSearchQueryOptions = {},
): string[] {
const maxQueries = options.maxQueries ?? 3;
const cleaned = cleanContent(content);
if (!cleaned) return [];
const candidates = new Map<string, PhraseCandidate>();
const addCandidate = (raw: string, bonus: number): void => {
const scored = scorePhrase(raw, bonus);
if (!scored) return;
const key = scored.phrase.toLowerCase();
const existing = candidates.get(key);
if (!existing || scored.score > existing.score) {
candidates.set(key, scored);
}
};
// 1. Quoted phrases — explicit user intent, strongest signal.
const quotedPhrases = cleaned.match(/"([^"]{2,80})"|'([^']{2,80})'/g);
if (quotedPhrases) { if (quotedPhrases) {
for (const phrase of quotedPhrases) { for (const phrase of quotedPhrases) {
const clean = phrase.replace(/["']/g, "").trim(); addCandidate(phrase.replace(/["']/g, ""), 10);
if (clean.length >= 3) queries.add(clean);
} }
} }
// 2. "nonton X" pattern — extract the title // 2. Definition/factual intent ("apa itu X", "arti X", "what is X").
const nontonMatch = content.match( const definitionMatch = cleaned.match(DEFINITION_INTENT);
/\b(nonton|tonton|rekomen|cari|search|google)\s+(.+?)(?:\s+(?:anime|kartun|film|movie|series|serial))?\s*[!?.]*$/i, if (definitionMatch) {
); addCandidate(definitionMatch[1].replace(/[?!.]+$/, ""), 8);
if (nontonMatch) {
const title = nontonMatch[2].trim();
if (title.length >= 2 && title.length <= 80) {
queries.add(title);
}
} }
// 3. "X anime/film" pattern — title before category // 3. "nonton/cari/rekomen/... X" — phrase after an intent verb.
const titleBeforeCategory = content.match( const intentMatch = cleaned.match(SEARCH_INTENT_VERBS);
/\b(\w[\w\s]{2,40})\s+(?:anime|kartun|film|movie|series|serial)\b/i, if (intentMatch) {
); addCandidate(intentMatch[1], 6);
if (titleBeforeCategory) {
const title = titleBeforeCategory[1].trim();
if (
title.length >= 3 &&
!/^(yang|yang|sama|dari|untuk|ini|itu|ada)$/i.test(title)
) {
queries.add(title);
}
} }
// 4. Standalone proper nouns (2+ words, capitalized) that look like titles // 4. Proper-noun phrases anywhere in the message — titles/named entities
const properNouns = content.match( // surface here even with no trigger verb.
/\b([A-Z][a-z]+(?:\s+[A-Z][a-z]+){1,4})\b/g, for (const m of cleaned.matchAll(PROPER_NOUN_PHRASE)) {
); addCandidate(m[1], 0);
if (properNouns) {
for (const noun of properNouns) {
// Skip common non-title proper nouns
const skip =
/^(Discord|YouTube|Google|Facebook|Instagram|Twitter|Github|ChatGPT|OpenAI|Claude|Telegram|WhatsApp|TikTok|Netflix|Spotify|Steam|Instagram)$/i;
if (!skip.test(noun) && noun.length >= 5) {
queries.add(noun);
}
}
} }
// 5. Terms that suggest research intent return Array.from(candidates.values())
const researchTerms = content.match( .sort((a, b) => b.score - a.score)
/\b(apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti|meaning|definisi|definition)\s+(.{3,60})/i, .slice(0, maxQueries)
); .map((c) => c.phrase);
if (researchTerms) {
const term = researchTerms[2].trim().replace(/[?!.]+$/, "");
if (term.length >= 3) queries.add(term);
}
return Array.from(queries).slice(0, 3);
} }
/** /**
@@ -0,0 +1,61 @@
// ═══════════════════════════════════════════════════════════════════════════
// extractSearchQueries — pure scoring-based extraction (no DB, Redis, or
// network). Replaces the old fixed-regex/category-list version (2026-08-31);
// these cases check the new scored extractor still covers what the old
// pattern list covered, plus the generalization gains.
// ═══════════════════════════════════════════════════════════════════════════
import { describe, expect, it } from "vitest";
import { extractSearchQueries } from "../src/modules/ai-moderation/wikipediaClient.js";
describe("extractSearchQueries", () => {
it("returns [] for plain conversational text with no lookup-worthy content", () => {
expect(extractSearchQueries("iya bener banget sih wkwkwk")).toEqual([]);
});
it("extracts a quoted phrase as the top candidate", () => {
const queries = extractSearchQueries('dia bilang "kostum hewan" itu aneh');
expect(queries[0]).toBe("kostum hewan");
});
it("extracts the target of a definition question without a fixed keyword list", () => {
const queries = extractSearchQueries("apa itu shirkmaxxing?");
expect(queries).toContain("shirkmaxxing");
});
it("extracts an intent-verb phrase with NO trailing category word required", () => {
// Old regex required a trailing anime|kartun|film|movie|series|serial to
// even try; the new version doesn't need one at all.
const queries = extractSearchQueries("woy nonton Attack on Titan dong");
expect(queries.some((q) => /attack on titan/i.test(q))).toBe(true);
});
it("extracts a multi-word proper-noun title with no trigger verb at all", () => {
const queries = extractSearchQueries("Chrono Cross itu keren banget");
expect(queries).toContain("Chrono Cross");
});
it("skips known-safe brand terms instead of relying on a fixed skip-list copy", () => {
// "Discord" is in the shared KNOWN_SAFE_TERMS set (textSignals.ts), reused
// here instead of a second hardcoded skip-list.
const queries = extractSearchQueries("Discord lagi down nih parah");
expect(queries).not.toContain("Discord");
});
it("caps results at maxQueries, highest-scored first", () => {
const queries = extractSearchQueries(
'nonton Xenogears sama "Chrono Cross" terus Yakuza juga',
{ maxQueries: 2 },
);
expect(queries.length).toBeLessThanOrEqual(2);
// Quoted phrase (bonus 10) should outrank the bare proper nouns.
expect(queries[0]).toBe("Chrono Cross");
});
it("strips URLs and mentions before extracting", () => {
const queries = extractSearchQueries(
'cek https://example.com/foo <@123456> "kafircel"',
);
expect(queries).toContain("kafircel");
expect(queries.some((q) => /example|123456/.test(q))).toBe(false);
});
});