feat(gateway): refactor term extraction and scoring logic into textSignals.ts for reuse
This commit is contained in:
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -22,6 +22,12 @@ import { createChildLogger } from "@/shared/logger/index";
|
|||||||
import { createAbortControllerWithTimeout } from "@/shared/utils/index";
|
import { createAbortControllerWithTimeout } from "@/shared/utils/index";
|
||||||
import { config } from "../../shared/config/config.js";
|
import { config } from "../../shared/config/config.js";
|
||||||
import { cacheGet, cacheSet, makeCacheKey } from "./cacheStore.js";
|
import { cacheGet, cacheSet, makeCacheKey } from "./cacheStore.js";
|
||||||
|
import {
|
||||||
|
cleanContent,
|
||||||
|
isKnownTerm,
|
||||||
|
isMostlyStopwords,
|
||||||
|
scoreWord,
|
||||||
|
} from "./textSignals.js";
|
||||||
|
|
||||||
const log = createChildLogger("wikipedia-client");
|
const log = createChildLogger("wikipedia-client");
|
||||||
|
|
||||||
@@ -206,73 +212,121 @@ export async function wikipediaSummary(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
export interface ExtractSearchQueryOptions {
|
||||||
* Extract meaningful search queries from message content.
|
maxQueries?: number;
|
||||||
* Uses multiple strategies to find terms worth searching.
|
}
|
||||||
* Returns up to 3 clean queries.
|
|
||||||
*/
|
|
||||||
export function extractSearchQueries(content: string): string[] {
|
|
||||||
const queries = new Set<string>();
|
|
||||||
|
|
||||||
// 1. Quoted phrases (explicit user intent)
|
/**
|
||||||
const quotedPhrases = content.match(/"([^"]+)"|'([^']+)'/g);
|
* Verbs that signal "find/watch/download X" — used only to find a phrase
|
||||||
|
* BOUNDARY. Unlike the old version, no trailing category word
|
||||||
|
* ("anime|film|series") is required, so coverage isn't capped by an
|
||||||
|
* enumerated category list. The capture stops at the first coordinating
|
||||||
|
* conjunction (or end of message) so "nonton X sama Y terus Z" yields the
|
||||||
|
* single entity X instead of swallowing the whole multi-entity tail into
|
||||||
|
* one unsearchable blob.
|
||||||
|
*/
|
||||||
|
const SEARCH_INTENT_VERBS =
|
||||||
|
/\b(?:nonton|tonton|rekomen(?:dasiin|dasikan)?|cari(?:in|kan)?|search|google|download|donlod|unduh|streaming|baca|dengerin|dengar(?:kan)?)\b\s+(.+?)(?=\s+(?:sama|dan|juga|terus|lalu|atau|and|or)\b|\s*[!?.]*$)/i;
|
||||||
|
|
||||||
|
/** "apa itu X" / "arti X" / "what is X" — factual/definition intent. */
|
||||||
|
const DEFINITION_INTENT =
|
||||||
|
/\b(?:apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti(?:nya)?|meaning(?:\s+of)?|definisi(?:nya)?|definition(?:\s+of)?)\s+(.{3,80})/i;
|
||||||
|
|
||||||
|
/** Multi-word capitalized runs — usually a title/named entity regardless of
|
||||||
|
* surrounding verbs (e.g. "Attack on Titan", "One Piece"). No trigger word
|
||||||
|
* needed at all. */
|
||||||
|
const PROPER_NOUN_PHRASE = /\b([A-Z][\p{L}]*(?:\s+[A-Z][\p{L}]*){1,4})\b/gu;
|
||||||
|
|
||||||
|
interface PhraseCandidate {
|
||||||
|
phrase: string;
|
||||||
|
score: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Scores a phrase with the SAME per-word signals as the term glossary
|
||||||
|
* (proper-noun casing, foreign spelling, hyphenation — see textSignals.ts),
|
||||||
|
* plus a bonus for the extraction strategy that surfaced it. Returns null
|
||||||
|
* for junk (mostly stopwords, or made entirely of known-safe terms like
|
||||||
|
* "Discord"/"Google" — those never need a Wikipedia lookup).
|
||||||
|
*/
|
||||||
|
function scorePhrase(rawPhrase: string, bonus: number): PhraseCandidate | null {
|
||||||
|
const clean = rawPhrase
|
||||||
|
.trim()
|
||||||
|
.replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "");
|
||||||
|
if (clean.length < 2 || clean.length > 80) return null;
|
||||||
|
if (isMostlyStopwords(clean)) return null;
|
||||||
|
|
||||||
|
const words = clean.split(/\s+/);
|
||||||
|
let score = bonus;
|
||||||
|
let hasNonStopword = false;
|
||||||
|
for (const w of words) {
|
||||||
|
if (isKnownTerm(w.toLowerCase())) continue;
|
||||||
|
hasNonStopword = true;
|
||||||
|
score += scoreWord(w);
|
||||||
|
}
|
||||||
|
if (!hasNonStopword) return null;
|
||||||
|
|
||||||
|
return { phrase: clean, score };
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Extracts candidate phrases worth searching on Wikipedia — scored by the
|
||||||
|
* same word-level signals as the term glossary, instead of requiring an
|
||||||
|
* exact match against a fixed, manually-maintained category/brand list.
|
||||||
|
* Pure CPU-side regex + scoring — no network or LLM call, so using this
|
||||||
|
* more broadly never adds AI requests.
|
||||||
|
*
|
||||||
|
* Returns up to `maxQueries` phrases (default 3), highest-scored first.
|
||||||
|
*/
|
||||||
|
export function extractSearchQueries(
|
||||||
|
content: string,
|
||||||
|
options: ExtractSearchQueryOptions = {},
|
||||||
|
): string[] {
|
||||||
|
const maxQueries = options.maxQueries ?? 3;
|
||||||
|
const cleaned = cleanContent(content);
|
||||||
|
if (!cleaned) return [];
|
||||||
|
|
||||||
|
const candidates = new Map<string, PhraseCandidate>();
|
||||||
|
const addCandidate = (raw: string, bonus: number): void => {
|
||||||
|
const scored = scorePhrase(raw, bonus);
|
||||||
|
if (!scored) return;
|
||||||
|
const key = scored.phrase.toLowerCase();
|
||||||
|
const existing = candidates.get(key);
|
||||||
|
if (!existing || scored.score > existing.score) {
|
||||||
|
candidates.set(key, scored);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// 1. Quoted phrases — explicit user intent, strongest signal.
|
||||||
|
const quotedPhrases = cleaned.match(/"([^"]{2,80})"|'([^']{2,80})'/g);
|
||||||
if (quotedPhrases) {
|
if (quotedPhrases) {
|
||||||
for (const phrase of quotedPhrases) {
|
for (const phrase of quotedPhrases) {
|
||||||
const clean = phrase.replace(/["']/g, "").trim();
|
addCandidate(phrase.replace(/["']/g, ""), 10);
|
||||||
if (clean.length >= 3) queries.add(clean);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// 2. "nonton X" pattern — extract the title
|
// 2. Definition/factual intent ("apa itu X", "arti X", "what is X").
|
||||||
const nontonMatch = content.match(
|
const definitionMatch = cleaned.match(DEFINITION_INTENT);
|
||||||
/\b(nonton|tonton|rekomen|cari|search|google)\s+(.+?)(?:\s+(?:anime|kartun|film|movie|series|serial))?\s*[!?.]*$/i,
|
if (definitionMatch) {
|
||||||
);
|
addCandidate(definitionMatch[1].replace(/[?!.]+$/, ""), 8);
|
||||||
if (nontonMatch) {
|
|
||||||
const title = nontonMatch[2].trim();
|
|
||||||
if (title.length >= 2 && title.length <= 80) {
|
|
||||||
queries.add(title);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// 3. "X anime/film" pattern — title before category
|
// 3. "nonton/cari/rekomen/... X" — phrase after an intent verb.
|
||||||
const titleBeforeCategory = content.match(
|
const intentMatch = cleaned.match(SEARCH_INTENT_VERBS);
|
||||||
/\b(\w[\w\s]{2,40})\s+(?:anime|kartun|film|movie|series|serial)\b/i,
|
if (intentMatch) {
|
||||||
);
|
addCandidate(intentMatch[1], 6);
|
||||||
if (titleBeforeCategory) {
|
|
||||||
const title = titleBeforeCategory[1].trim();
|
|
||||||
if (
|
|
||||||
title.length >= 3 &&
|
|
||||||
!/^(yang|yang|sama|dari|untuk|ini|itu|ada)$/i.test(title)
|
|
||||||
) {
|
|
||||||
queries.add(title);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// 4. Standalone proper nouns (2+ words, capitalized) that look like titles
|
// 4. Proper-noun phrases anywhere in the message — titles/named entities
|
||||||
const properNouns = content.match(
|
// surface here even with no trigger verb.
|
||||||
/\b([A-Z][a-z]+(?:\s+[A-Z][a-z]+){1,4})\b/g,
|
for (const m of cleaned.matchAll(PROPER_NOUN_PHRASE)) {
|
||||||
);
|
addCandidate(m[1], 0);
|
||||||
if (properNouns) {
|
|
||||||
for (const noun of properNouns) {
|
|
||||||
// Skip common non-title proper nouns
|
|
||||||
const skip =
|
|
||||||
/^(Discord|YouTube|Google|Facebook|Instagram|Twitter|Github|ChatGPT|OpenAI|Claude|Telegram|WhatsApp|TikTok|Netflix|Spotify|Steam|Instagram)$/i;
|
|
||||||
if (!skip.test(noun) && noun.length >= 5) {
|
|
||||||
queries.add(noun);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// 5. Terms that suggest research intent
|
return Array.from(candidates.values())
|
||||||
const researchTerms = content.match(
|
.sort((a, b) => b.score - a.score)
|
||||||
/\b(apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti|meaning|definisi|definition)\s+(.{3,60})/i,
|
.slice(0, maxQueries)
|
||||||
);
|
.map((c) => c.phrase);
|
||||||
if (researchTerms) {
|
|
||||||
const term = researchTerms[2].trim().replace(/[?!.]+$/, "");
|
|
||||||
if (term.length >= 3) queries.add(term);
|
|
||||||
}
|
|
||||||
|
|
||||||
return Array.from(queries).slice(0, 3);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -0,0 +1,61 @@
|
|||||||
|
// ═══════════════════════════════════════════════════════════════════════════
|
||||||
|
// extractSearchQueries — pure scoring-based extraction (no DB, Redis, or
|
||||||
|
// network). Replaces the old fixed-regex/category-list version (2026-08-31);
|
||||||
|
// these cases check the new scored extractor still covers what the old
|
||||||
|
// pattern list covered, plus the generalization gains.
|
||||||
|
// ═══════════════════════════════════════════════════════════════════════════
|
||||||
|
import { describe, expect, it } from "vitest";
|
||||||
|
import { extractSearchQueries } from "../src/modules/ai-moderation/wikipediaClient.js";
|
||||||
|
|
||||||
|
describe("extractSearchQueries", () => {
|
||||||
|
it("returns [] for plain conversational text with no lookup-worthy content", () => {
|
||||||
|
expect(extractSearchQueries("iya bener banget sih wkwkwk")).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("extracts a quoted phrase as the top candidate", () => {
|
||||||
|
const queries = extractSearchQueries('dia bilang "kostum hewan" itu aneh');
|
||||||
|
expect(queries[0]).toBe("kostum hewan");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("extracts the target of a definition question without a fixed keyword list", () => {
|
||||||
|
const queries = extractSearchQueries("apa itu shirkmaxxing?");
|
||||||
|
expect(queries).toContain("shirkmaxxing");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("extracts an intent-verb phrase with NO trailing category word required", () => {
|
||||||
|
// Old regex required a trailing anime|kartun|film|movie|series|serial to
|
||||||
|
// even try; the new version doesn't need one at all.
|
||||||
|
const queries = extractSearchQueries("woy nonton Attack on Titan dong");
|
||||||
|
expect(queries.some((q) => /attack on titan/i.test(q))).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("extracts a multi-word proper-noun title with no trigger verb at all", () => {
|
||||||
|
const queries = extractSearchQueries("Chrono Cross itu keren banget");
|
||||||
|
expect(queries).toContain("Chrono Cross");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("skips known-safe brand terms instead of relying on a fixed skip-list copy", () => {
|
||||||
|
// "Discord" is in the shared KNOWN_SAFE_TERMS set (textSignals.ts), reused
|
||||||
|
// here instead of a second hardcoded skip-list.
|
||||||
|
const queries = extractSearchQueries("Discord lagi down nih parah");
|
||||||
|
expect(queries).not.toContain("Discord");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("caps results at maxQueries, highest-scored first", () => {
|
||||||
|
const queries = extractSearchQueries(
|
||||||
|
'nonton Xenogears sama "Chrono Cross" terus Yakuza juga',
|
||||||
|
{ maxQueries: 2 },
|
||||||
|
);
|
||||||
|
expect(queries.length).toBeLessThanOrEqual(2);
|
||||||
|
// Quoted phrase (bonus 10) should outrank the bare proper nouns.
|
||||||
|
expect(queries[0]).toBe("Chrono Cross");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("strips URLs and mentions before extracting", () => {
|
||||||
|
const queries = extractSearchQueries(
|
||||||
|
'cek https://example.com/foo <@123456> "kafircel"',
|
||||||
|
);
|
||||||
|
expect(queries).toContain("kafircel");
|
||||||
|
expect(queries.some((q) => /example|123456/.test(q))).toBe(false);
|
||||||
|
});
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user