From 0e31aa06b873cc8fbcdbedf7d857f83791b6d718 Mon Sep 17 00:00:00 2001 From: asepharyana Date: Mon, 31 Aug 2026 22:59:23 +0700 Subject: [PATCH] feat(gateway): refactor term extraction and scoring logic into textSignals.ts for reuse --- .../src/modules/ai-moderation/termGlossary.ts | 106 ++--------- .../src/modules/ai-moderation/textSignals.ts | 106 +++++++++++ .../modules/ai-moderation/wikipediaClient.ts | 164 ++++++++++++------ .../tests/wikipediaSearchQueries.test.ts | 61 +++++++ 4 files changed, 288 insertions(+), 149 deletions(-) create mode 100644 services/discord-gateway/src/modules/ai-moderation/textSignals.ts create mode 100644 services/discord-gateway/tests/wikipediaSearchQueries.test.ts diff --git a/services/discord-gateway/src/modules/ai-moderation/termGlossary.ts b/services/discord-gateway/src/modules/ai-moderation/termGlossary.ts index f2d0988f..82726aa8 100644 --- a/services/discord-gateway/src/modules/ai-moderation/termGlossary.ts +++ b/services/discord-gateway/src/modules/ai-moderation/termGlossary.ts @@ -41,6 +41,14 @@ import { getTermDefinitionFromDb, setTermDefinitionInDb, } from "./termGlossaryStore.js"; +import { + cleanContent, + isKnownTerm, + isMostlyStopwords, + isNoiseWord, + scoreWord, + WORD_RE, +} from "./textSignals.js"; import { wikipediaSummary } from "./wikipediaClient.js"; const log = createChildLogger("term-glossary"); @@ -96,100 +104,10 @@ async function acquireLiveSlot(): Promise { // --------------------------------------------------------------------------- // Term extraction // --------------------------------------------------------------------------- - -/** Word tokenizer — letters/digits plus internal -_'· (handles "well-known", - * "node_modules", diacritics). */ -const WORD_RE = /[\p{L}\p{N}]+(?:[-_'’·][\p{L}\p{N}]+)*/gu; - -/** Removes URLs, Discord mentions/custom emoji, code fences, markdown noise. */ -function cleanContent(raw: string): string { - return raw - .replace(/https?:\/\/\S+/gi, " ") - .replace(/<@!?\d+>/g, " ") - .replace(/<#\d+>/g, " ") - .replace(//g, " ") - .replace(/[`*_~|>[\]]/g, " ") - .replace(/[\p{Emoji}\p{Extended_Pictographic}]/gu, " ") - .replace(/\s+/g, " ") - .trim(); -} - -/** Filters out tokens that are useless as glossary candidates (numbers, - * repeated-char noise, mega-tokens). */ -function isNoiseWord(word: string): boolean { - if (word.length > 28) return true; - if (/^\d+$/.test(word)) return true; - const lower = word.toLowerCase(); - // "aaaa…", "wwwwww" — single repeated character - if (/^(.)\1{2,}$/.test(lower)) return true; - // "wkwk", "hehe", "69" alternations — repeated 2–3 char base. "meme" is - // the one legit 4-letter word this matches; it is whitelisted below. - if (/^([a-z]{2,3})\1{1,}$/.test(lower)) return true; - return false; -} - -/** Deterministic bonus for words that look like proper nouns or foreign. */ -function scoreWord(word: string): number { - let score = 1; - // Capitalized first letter (proper noun / title) but not ALL-CAPS acronyms - if (/^[A-Z]/.test(word) && !/^[A-Z]{2,}$/.test(word)) score += 3; - // Contains a letter outside basic latin → regional/foreign spelling - if (/[\p{L}]/u.test(word.replace(/[A-Za-z]/g, ""))) score += 2; - // Contains an internal apostrophe or hyphen → likely a named entity - if (/[-_'’·]/.test(word)) score += 2; - return score; -} - -const STOPWORDS = new Set( - // ── Bahasa Indonesia ──────────────────────────────────────────────── - ( - " yang dan di ke dari ini itu dengan untuk pada dalam adalah akan telah sudah bisa dapat harus tidak juga saya kamu kita kami mereka dia aku kau gua lu lo gw gue elu anda kalian nya kah lah pun ya yah kan sih dong deh kok loh toh aja saja gitu gini begitu begini tapi tetapi namun atau karena sebab jika kalau bila maka supaya agar meski meskipun walau walaupun ketika saat setelah sebelum selama antara terhadap tentang mengenai bagi oleh secara sebagai seperti daripada tanpa hingga sampai sejak menuju bahwa padahal sebenarnya sepertinya mungkin memang jadi lalu terus akhirnya misalnya contohnya banyak sedikit semua seluruh setiap tiap beberapa ada bukan jangan boleh mau ingin pengen nggak ngak gak ga kagak ngga ndak nanti kemarin besok hari ini sekarang waktu itu masih sedang belum pernah sering selalu kadang jarang cepat lambat awal akhir baru lama besar kecil tinggi rendah panjang pendek baik buruk benar salah sama beda penting biasanya selamat terima kasih makasih sangat sekali paling cuma cuman hanya lebih kurang sekitar hampir ternyata rupanya begitu gimana bagaimana kenapa mengapa siapa apa mana kapan darimana kemana bilang ngomong omong kata tadi dulu terus lagi tetap pasti seharusnya sebaiknya seakan seolah kayaknya keliatan kelihatan ketahuan disini disitu disana kesini kesana bener pake pakai kayak emang lagian mulu istilah istilahnya banget" + - // ── English ─────────────────────────────────────────────────────── - " the a an and or but if then else for to in on at by with without from of is are was were be been being have has had do does did will would can could should may might must shall this that these those it its i you he she we they them their there here when where why how what which who whom whose only very just about above after before below under over into onto within upon against between among during through across along around behind beyond near off out up down now then so as not no yes ok okay" + - // ── Common net slang / acronyms the LLM already knows ────────────── - " lol omg wtf idk btw tbh imo aka fyi nsfw smh nvm asap afk brb gg wp ty np mb sry thx kk oke okk ygy frfr" - ).split(/\s+/), -); - -/** - * Words that are either already defined by the moderation rules, or are so - * common (brands, tech vocabulary, project names) that a Wikipedia lookup is - * a guaranteed miss/waste. Keeps the glossary focused on genuinely unknown - * terms. - */ -const KNOWN_SAFE_TERMS = new Set( - ( - "discord youtube google facebook instagram twitter tiktok whatsapp telegram netflix spotify steam github gitlab bitbucket chatgpt openai anthropic claude deepseek gemini llama copilot cursor vscode vscodium jetbrains intellij pycharm webstorm sublime codeblocks" + - " docker kubernetes k8s linux ubuntu debian arch fedora manjaro kali windows macos android ios chrome firefox safari edge opera brave" + - " react nextjs next vue svelte angular node nodejs deno bun pnpm yarn npm javascript typescript python golang go rust java kotlin swift cplusplus cpp css html json xml yaml toml regex backend frontend database mysql postgres postgresql mongodb redis qdrant sqlite nosql graphql rest websocket webhook" + - " bug crash error debug fix issue pr merge commit push pull branch main master dev staging production server client app website web browser" + - " stream streaming video audio voice call camera screen share screenshare gameplay gaming game play steam epic xbox playstation nintendo switch console" + - " bot discordbot moderation moderator admin member user profile avatar channel server guild message chat dm reply forward embed sticker emoji role permission" + - " meme code coding ngoding programmer program developer engineer software hardware cpu gpu ram rom storage disk network internet wifi lan ip dns vpn proxy cloud aws azure gcp vercel netlify heroku railway render vps hosting domain ssl login logout register account password email username" + - " anime manga waifu husbando tsundere moe otaku wibu weeb otome isekai shonen seinen josei manga manhwa manhua doujin" + - " anjay wkwk wkwkwk gws gaskeun santuy njir baka woy woi hadeh astaga asu anjing bangsat ngehe asal alay lebay caper mabar" + - " asus bete imphnen impnhen ngab" + - " syahadat sholat shalat solat puasa zakat haji umrah doa tuhan nabi allah yesus muhammad hashem" + - " loli shota incest exhibition furry fursuit cosplay costume" + - " gaza palestine israel yahudi yahud israel palestina israeli" + - " hokkian mandarin arabic jawa sunda betawi minang bugis batak melayu inggris indonesia" - ).split(/\s+/), -); - -function isKnownTerm(word: string): boolean { - return STOPWORDS.has(word) || KNOWN_SAFE_TERMS.has(word); -} - -/** True when a quoted phrase is mostly filler words (skip it). */ -function isMostlyStopwords(phrase: string): boolean { - const words = phrase - .toLowerCase() - .split(/[^a-zà-öø-ÿ]+/i) - .filter(Boolean); - if (words.length === 0) return true; - const stopCount = words.filter((w) => STOPWORDS.has(w)).length; - return stopCount / words.length >= 0.6; -} +// Tokenizer, stopword/known-term filters, and word scoring now live in +// textSignals.ts (2026-08-31) — shared with wikipediaClient.ts's search-query +// extractor so both use the exact same "is this word worth looking up" +// signals instead of two divergent copies. export interface ExtractGlossaryOptions { maxTerms?: number; diff --git a/services/discord-gateway/src/modules/ai-moderation/textSignals.ts b/services/discord-gateway/src/modules/ai-moderation/textSignals.ts new file mode 100644 index 00000000..dc76c489 --- /dev/null +++ b/services/discord-gateway/src/modules/ai-moderation/textSignals.ts @@ -0,0 +1,106 @@ +/** + * textSignals.ts + * + * Shared, LLM-free text scoring signals for deciding which words/phrases in + * a message are worth an external lookup (Wikipedia glossary term, Wikipedia + * search query, etc.). + * + * Extracted out of termGlossary.ts (2026-08-31) so wikipediaClient.ts's + * phrase-level search-query extractor can reuse the exact same + * stopword/known-term/scoring logic instead of maintaining a second, + * divergent copy. Everything here is pure CPU-side regex + scoring — no + * network or LLM call — so using it in more places never adds AI requests. + */ + +/** Word tokenizer — letters/digits plus internal -_'· (handles "well-known", + * "node_modules", diacritics). */ +export const WORD_RE = /[\p{L}\p{N}]+(?:[-_'’·][\p{L}\p{N}]+)*/gu; + +/** Removes URLs, Discord mentions/custom emoji, code fences, markdown noise. */ +export function cleanContent(raw: string): string { + return raw + .replace(/https?:\/\/\S+/gi, " ") + .replace(/<@!?\d+>/g, " ") + .replace(/<#\d+>/g, " ") + .replace(//g, " ") + .replace(/[`*_~|>[\]]/g, " ") + .replace(/[\p{Emoji}\p{Extended_Pictographic}]/gu, " ") + .replace(/\s+/g, " ") + .trim(); +} + +/** Filters out tokens that are useless as lookup candidates (numbers, + * repeated-char noise, mega-tokens). */ +export function isNoiseWord(word: string): boolean { + if (word.length > 28) return true; + if (/^\d+$/.test(word)) return true; + const lower = word.toLowerCase(); + // "aaaa…", "wwwwww" — single repeated character + if (/^(.)\1{2,}$/.test(lower)) return true; + // "wkwk", "hehe", "69" alternations — repeated 2–3 char base. "meme" is + // the one legit 4-letter word this matches; it is whitelisted below. + if (/^([a-z]{2,3})\1{1,}$/.test(lower)) return true; + return false; +} + +/** Deterministic bonus for words that look like proper nouns or foreign. */ +export function scoreWord(word: string): number { + let score = 1; + // Capitalized first letter (proper noun / title) but not ALL-CAPS acronyms + if (/^[A-Z]/.test(word) && !/^[A-Z]{2,}$/.test(word)) score += 3; + // Contains a letter outside basic latin → regional/foreign spelling + if (/[\p{L}]/u.test(word.replace(/[A-Za-z]/g, ""))) score += 2; + // Contains an internal apostrophe or hyphen → likely a named entity + if (/[-_'’·]/.test(word)) score += 2; + return score; +} + +export const STOPWORDS = new Set( + // ── Bahasa Indonesia ──────────────────────────────────────────────── + ( + " yang dan di ke dari ini itu dengan untuk pada dalam adalah akan telah sudah bisa dapat harus tidak juga saya kamu kita kami mereka dia aku kau gua lu lo gw gue elu anda kalian nya kah lah pun ya yah kan sih dong deh kok loh toh aja saja gitu gini begitu begini tapi tetapi namun atau karena sebab jika kalau bila maka supaya agar meski meskipun walau walaupun ketika saat setelah sebelum selama antara terhadap tentang mengenai bagi oleh secara sebagai seperti daripada tanpa hingga sampai sejak menuju bahwa padahal sebenarnya sepertinya mungkin memang jadi lalu terus akhirnya misalnya contohnya banyak sedikit semua seluruh setiap tiap beberapa ada bukan jangan boleh mau ingin pengen nggak ngak gak ga kagak ngga ndak nanti kemarin besok hari ini sekarang waktu itu masih sedang belum pernah sering selalu kadang jarang cepat lambat awal akhir baru lama besar kecil tinggi rendah panjang pendek baik buruk benar salah sama beda penting biasanya selamat terima kasih makasih sangat sekali paling cuma cuman hanya lebih kurang sekitar hampir ternyata rupanya begitu gimana bagaimana kenapa mengapa siapa apa mana kapan darimana kemana bilang ngomong omong kata tadi dulu terus lagi tetap pasti seharusnya sebaiknya seakan seolah kayaknya keliatan kelihatan ketahuan disini disitu disana kesini kesana bener pake pakai kayak emang lagian mulu istilah istilahnya banget" + + // ── English ─────────────────────────────────────────────────────── + " the a an and or but if then else for to in on at by with without from of is are was were be been being have has had do does did will would can could should may might must shall this that these those it its i you he she we they them their there here when where why how what which who whom whose only very just about above after before below under over into onto within upon against between among during through across along around behind beyond near off out up down now then so as not no yes ok okay" + + // ── Common net slang / acronyms the LLM already knows ────────────── + " lol omg wtf idk btw tbh imo aka fyi nsfw smh nvm asap afk brb gg wp ty np mb sry thx kk oke okk ygy frfr" + ).split(/\s+/), +); + +/** + * Words that are either already defined by the moderation rules, or are so + * common (brands, tech vocabulary, project names) that a Wikipedia lookup is + * a guaranteed miss/waste. Keeps lookups focused on genuinely unknown terms. + */ +export const KNOWN_SAFE_TERMS = new Set( + ( + "discord youtube google facebook instagram twitter tiktok whatsapp telegram netflix spotify steam github gitlab bitbucket chatgpt openai anthropic claude deepseek gemini llama copilot cursor vscode vscodium jetbrains intellij pycharm webstorm sublime codeblocks" + + " docker kubernetes k8s linux ubuntu debian arch fedora manjaro kali windows macos android ios chrome firefox safari edge opera brave" + + " react nextjs next vue svelte angular node nodejs deno bun pnpm yarn npm javascript typescript python golang go rust java kotlin swift cplusplus cpp css html json xml yaml toml regex backend frontend database mysql postgres postgresql mongodb redis qdrant sqlite nosql graphql rest websocket webhook" + + " bug crash error debug fix issue pr merge commit push pull branch main master dev staging production server client app website web browser" + + " stream streaming video audio voice call camera screen share screenshare gameplay gaming game play steam epic xbox playstation nintendo switch console" + + " bot discordbot moderation moderator admin member user profile avatar channel server guild message chat dm reply forward embed sticker emoji role permission" + + " meme code coding ngoding programmer program developer engineer software hardware cpu gpu ram rom storage disk network internet wifi lan ip dns vpn proxy cloud aws azure gcp vercel netlify heroku railway render vps hosting domain ssl login logout register account password email username" + + " anime manga waifu husbando tsundere moe otaku wibu weeb otome isekai shonen seinen josei manga manhwa manhua doujin" + + " anjay wkwk wkwkwk gws gaskeun santuy njir baka woy woi hadeh astaga asu anjing bangsat ngehe asal alay lebay caper mabar" + + " asus bete imphnen impnhen ngab" + + " syahadat sholat shalat solat puasa zakat haji umrah doa tuhan nabi allah yesus muhammad hashem" + + " loli shota incest exhibition furry fursuit cosplay costume" + + " gaza palestine israel yahudi yahud israel palestina israeli" + + " hokkian mandarin arabic jawa sunda betawi minang bugis batak melayu inggris indonesia" + ).split(/\s+/), +); + +export function isKnownTerm(word: string): boolean { + return STOPWORDS.has(word) || KNOWN_SAFE_TERMS.has(word); +} + +/** True when a phrase is mostly filler words (skip it as a lookup candidate). */ +export function isMostlyStopwords(phrase: string): boolean { + const words = phrase + .toLowerCase() + .split(/[^a-zà-öø-ÿ]+/i) + .filter(Boolean); + if (words.length === 0) return true; + const stopCount = words.filter((w) => STOPWORDS.has(w)).length; + return stopCount / words.length >= 0.6; +} diff --git a/services/discord-gateway/src/modules/ai-moderation/wikipediaClient.ts b/services/discord-gateway/src/modules/ai-moderation/wikipediaClient.ts index 41b21bc0..1442c914 100644 --- a/services/discord-gateway/src/modules/ai-moderation/wikipediaClient.ts +++ b/services/discord-gateway/src/modules/ai-moderation/wikipediaClient.ts @@ -22,6 +22,12 @@ import { createChildLogger } from "@/shared/logger/index"; import { createAbortControllerWithTimeout } from "@/shared/utils/index"; import { config } from "../../shared/config/config.js"; import { cacheGet, cacheSet, makeCacheKey } from "./cacheStore.js"; +import { + cleanContent, + isKnownTerm, + isMostlyStopwords, + scoreWord, +} from "./textSignals.js"; const log = createChildLogger("wikipedia-client"); @@ -206,73 +212,121 @@ export async function wikipediaSummary( } } -/** - * Extract meaningful search queries from message content. - * Uses multiple strategies to find terms worth searching. - * Returns up to 3 clean queries. - */ -export function extractSearchQueries(content: string): string[] { - const queries = new Set(); +export interface ExtractSearchQueryOptions { + maxQueries?: number; +} - // 1. Quoted phrases (explicit user intent) - const quotedPhrases = content.match(/"([^"]+)"|'([^']+)'/g); +/** + * Verbs that signal "find/watch/download X" — used only to find a phrase + * BOUNDARY. Unlike the old version, no trailing category word + * ("anime|film|series") is required, so coverage isn't capped by an + * enumerated category list. The capture stops at the first coordinating + * conjunction (or end of message) so "nonton X sama Y terus Z" yields the + * single entity X instead of swallowing the whole multi-entity tail into + * one unsearchable blob. + */ +const SEARCH_INTENT_VERBS = + /\b(?:nonton|tonton|rekomen(?:dasiin|dasikan)?|cari(?:in|kan)?|search|google|download|donlod|unduh|streaming|baca|dengerin|dengar(?:kan)?)\b\s+(.+?)(?=\s+(?:sama|dan|juga|terus|lalu|atau|and|or)\b|\s*[!?.]*$)/i; + +/** "apa itu X" / "arti X" / "what is X" — factual/definition intent. */ +const DEFINITION_INTENT = + /\b(?:apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti(?:nya)?|meaning(?:\s+of)?|definisi(?:nya)?|definition(?:\s+of)?)\s+(.{3,80})/i; + +/** Multi-word capitalized runs — usually a title/named entity regardless of + * surrounding verbs (e.g. "Attack on Titan", "One Piece"). No trigger word + * needed at all. */ +const PROPER_NOUN_PHRASE = /\b([A-Z][\p{L}]*(?:\s+[A-Z][\p{L}]*){1,4})\b/gu; + +interface PhraseCandidate { + phrase: string; + score: number; +} + +/** + * Scores a phrase with the SAME per-word signals as the term glossary + * (proper-noun casing, foreign spelling, hyphenation — see textSignals.ts), + * plus a bonus for the extraction strategy that surfaced it. Returns null + * for junk (mostly stopwords, or made entirely of known-safe terms like + * "Discord"/"Google" — those never need a Wikipedia lookup). + */ +function scorePhrase(rawPhrase: string, bonus: number): PhraseCandidate | null { + const clean = rawPhrase + .trim() + .replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, ""); + if (clean.length < 2 || clean.length > 80) return null; + if (isMostlyStopwords(clean)) return null; + + const words = clean.split(/\s+/); + let score = bonus; + let hasNonStopword = false; + for (const w of words) { + if (isKnownTerm(w.toLowerCase())) continue; + hasNonStopword = true; + score += scoreWord(w); + } + if (!hasNonStopword) return null; + + return { phrase: clean, score }; +} + +/** + * Extracts candidate phrases worth searching on Wikipedia — scored by the + * same word-level signals as the term glossary, instead of requiring an + * exact match against a fixed, manually-maintained category/brand list. + * Pure CPU-side regex + scoring — no network or LLM call, so using this + * more broadly never adds AI requests. + * + * Returns up to `maxQueries` phrases (default 3), highest-scored first. + */ +export function extractSearchQueries( + content: string, + options: ExtractSearchQueryOptions = {}, +): string[] { + const maxQueries = options.maxQueries ?? 3; + const cleaned = cleanContent(content); + if (!cleaned) return []; + + const candidates = new Map(); + const addCandidate = (raw: string, bonus: number): void => { + const scored = scorePhrase(raw, bonus); + if (!scored) return; + const key = scored.phrase.toLowerCase(); + const existing = candidates.get(key); + if (!existing || scored.score > existing.score) { + candidates.set(key, scored); + } + }; + + // 1. Quoted phrases — explicit user intent, strongest signal. + const quotedPhrases = cleaned.match(/"([^"]{2,80})"|'([^']{2,80})'/g); if (quotedPhrases) { for (const phrase of quotedPhrases) { - const clean = phrase.replace(/["']/g, "").trim(); - if (clean.length >= 3) queries.add(clean); + addCandidate(phrase.replace(/["']/g, ""), 10); } } - // 2. "nonton X" pattern — extract the title - const nontonMatch = content.match( - /\b(nonton|tonton|rekomen|cari|search|google)\s+(.+?)(?:\s+(?:anime|kartun|film|movie|series|serial))?\s*[!?.]*$/i, - ); - if (nontonMatch) { - const title = nontonMatch[2].trim(); - if (title.length >= 2 && title.length <= 80) { - queries.add(title); - } + // 2. Definition/factual intent ("apa itu X", "arti X", "what is X"). + const definitionMatch = cleaned.match(DEFINITION_INTENT); + if (definitionMatch) { + addCandidate(definitionMatch[1].replace(/[?!.]+$/, ""), 8); } - // 3. "X anime/film" pattern — title before category - const titleBeforeCategory = content.match( - /\b(\w[\w\s]{2,40})\s+(?:anime|kartun|film|movie|series|serial)\b/i, - ); - if (titleBeforeCategory) { - const title = titleBeforeCategory[1].trim(); - if ( - title.length >= 3 && - !/^(yang|yang|sama|dari|untuk|ini|itu|ada)$/i.test(title) - ) { - queries.add(title); - } + // 3. "nonton/cari/rekomen/... X" — phrase after an intent verb. + const intentMatch = cleaned.match(SEARCH_INTENT_VERBS); + if (intentMatch) { + addCandidate(intentMatch[1], 6); } - // 4. Standalone proper nouns (2+ words, capitalized) that look like titles - const properNouns = content.match( - /\b([A-Z][a-z]+(?:\s+[A-Z][a-z]+){1,4})\b/g, - ); - if (properNouns) { - for (const noun of properNouns) { - // Skip common non-title proper nouns - const skip = - /^(Discord|YouTube|Google|Facebook|Instagram|Twitter|Github|ChatGPT|OpenAI|Claude|Telegram|WhatsApp|TikTok|Netflix|Spotify|Steam|Instagram)$/i; - if (!skip.test(noun) && noun.length >= 5) { - queries.add(noun); - } - } + // 4. Proper-noun phrases anywhere in the message — titles/named entities + // surface here even with no trigger verb. + for (const m of cleaned.matchAll(PROPER_NOUN_PHRASE)) { + addCandidate(m[1], 0); } - // 5. Terms that suggest research intent - const researchTerms = content.match( - /\b(apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti|meaning|definisi|definition)\s+(.{3,60})/i, - ); - if (researchTerms) { - const term = researchTerms[2].trim().replace(/[?!.]+$/, ""); - if (term.length >= 3) queries.add(term); - } - - return Array.from(queries).slice(0, 3); + return Array.from(candidates.values()) + .sort((a, b) => b.score - a.score) + .slice(0, maxQueries) + .map((c) => c.phrase); } /** diff --git a/services/discord-gateway/tests/wikipediaSearchQueries.test.ts b/services/discord-gateway/tests/wikipediaSearchQueries.test.ts new file mode 100644 index 00000000..052a8bf4 --- /dev/null +++ b/services/discord-gateway/tests/wikipediaSearchQueries.test.ts @@ -0,0 +1,61 @@ +// ═══════════════════════════════════════════════════════════════════════════ +// extractSearchQueries — pure scoring-based extraction (no DB, Redis, or +// network). Replaces the old fixed-regex/category-list version (2026-08-31); +// these cases check the new scored extractor still covers what the old +// pattern list covered, plus the generalization gains. +// ═══════════════════════════════════════════════════════════════════════════ +import { describe, expect, it } from "vitest"; +import { extractSearchQueries } from "../src/modules/ai-moderation/wikipediaClient.js"; + +describe("extractSearchQueries", () => { + it("returns [] for plain conversational text with no lookup-worthy content", () => { + expect(extractSearchQueries("iya bener banget sih wkwkwk")).toEqual([]); + }); + + it("extracts a quoted phrase as the top candidate", () => { + const queries = extractSearchQueries('dia bilang "kostum hewan" itu aneh'); + expect(queries[0]).toBe("kostum hewan"); + }); + + it("extracts the target of a definition question without a fixed keyword list", () => { + const queries = extractSearchQueries("apa itu shirkmaxxing?"); + expect(queries).toContain("shirkmaxxing"); + }); + + it("extracts an intent-verb phrase with NO trailing category word required", () => { + // Old regex required a trailing anime|kartun|film|movie|series|serial to + // even try; the new version doesn't need one at all. + const queries = extractSearchQueries("woy nonton Attack on Titan dong"); + expect(queries.some((q) => /attack on titan/i.test(q))).toBe(true); + }); + + it("extracts a multi-word proper-noun title with no trigger verb at all", () => { + const queries = extractSearchQueries("Chrono Cross itu keren banget"); + expect(queries).toContain("Chrono Cross"); + }); + + it("skips known-safe brand terms instead of relying on a fixed skip-list copy", () => { + // "Discord" is in the shared KNOWN_SAFE_TERMS set (textSignals.ts), reused + // here instead of a second hardcoded skip-list. + const queries = extractSearchQueries("Discord lagi down nih parah"); + expect(queries).not.toContain("Discord"); + }); + + it("caps results at maxQueries, highest-scored first", () => { + const queries = extractSearchQueries( + 'nonton Xenogears sama "Chrono Cross" terus Yakuza juga', + { maxQueries: 2 }, + ); + expect(queries.length).toBeLessThanOrEqual(2); + // Quoted phrase (bonus 10) should outrank the bare proper nouns. + expect(queries[0]).toBe("Chrono Cross"); + }); + + it("strips URLs and mentions before extracting", () => { + const queries = extractSearchQueries( + 'cek https://example.com/foo <@123456> "kafircel"', + ); + expect(queries).toContain("kafircel"); + expect(queries.some((q) => /example|123456/.test(q))).toBe(false); + }); +});