From e6e44b30c7926d8ec20b67bf43375d5218de7e51 Mon Sep 17 00:00:00 2001 From: MythEclipse Date: Sat, 6 Jun 2026 13:27:24 +0700 Subject: [PATCH] fix(ai-moderation): remove indonesianTextNormalizer to stop hallucinated slang flags The rule-based badword detector was injecting [normalized_text] and [normalization_notes] tags into the LLM prompt that caused false positive hallucinations - the LLM started associating innocent words ('sapik', 'furina') with furry/sexual_deviation due to misleading context injected by the normalizer. Removed: - indonesianTextNormalizer.ts (full file deletion) - formatModerationTextEvidenceForPrompt import/usage in llmModerationClient - formatModerationTextEvidenceForPrompt import/usage in conversationContext - stale re-exports in index.ts Co-Authored-By: Claude Opus 4.8 --- .../ai-moderation/conversationContext.ts | 5 +- .../src/modules/ai-moderation/index.ts | 5 - .../ai-moderation/indonesianTextNormalizer.ts | 430 ------------------ .../ai-moderation/llmModerationClient.ts | 21 +- 4 files changed, 4 insertions(+), 457 deletions(-) delete mode 100644 services/discord-gateway/src/modules/ai-moderation/indonesianTextNormalizer.ts diff --git a/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts b/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts index 4574b79..b4476c9 100644 --- a/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts +++ b/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts @@ -1,7 +1,6 @@ import { encoding_for_model as encodingForModel } from "tiktoken"; import { formatMediaEvidenceForPrompt } from "../message-capture/messageMetadata.js"; import type { MessageRecord } from "../message-capture/types.js"; -import { formatModerationTextEvidenceForPrompt } from "./indonesianTextNormalizer.js"; export interface ConversationContextInput { contextBefore: MessageRecord[]; @@ -42,11 +41,9 @@ export function formatMessageForPrompt( ): string { const content = msg.edited_content ?? msg.content; const timestamp = formatTimestamp(msg.created_at); - const textEvidence = formatModerationTextEvidenceForPrompt(content); - const textSuffix = textEvidence ? ` ${textEvidence}` : ""; const mediaEvidence = formatMediaEvidenceForPrompt(msg.metadata); const mediaSuffix = mediaEvidence ? ` ${mediaEvidence}` : ""; - return `[${label}] id=${msg.id} time=${timestamp} user=${msg.username}: ${content}${textSuffix}${mediaSuffix}`; + return `[${label}] id=${msg.id} time=${timestamp} user=${msg.username}: ${content}${mediaSuffix}`; } /** diff --git a/services/discord-gateway/src/modules/ai-moderation/index.ts b/services/discord-gateway/src/modules/ai-moderation/index.ts index 584ad99..7d93daa 100644 --- a/services/discord-gateway/src/modules/ai-moderation/index.ts +++ b/services/discord-gateway/src/modules/ai-moderation/index.ts @@ -1,8 +1,3 @@ export { startPendingAIAnalysisWorker } from "./aiAnalyzer.js"; -export { - buildModerationTextEvidence, - detectIndonesianBadwords, - normalizeDiscordCustomEmoji, -} from "./indonesianTextNormalizer.js"; export { runModerationAnalysis } from "./llmModerationClient.js"; export { buildSystemPrompt } from "./moderationPrompt.js"; diff --git a/services/discord-gateway/src/modules/ai-moderation/indonesianTextNormalizer.ts b/services/discord-gateway/src/modules/ai-moderation/indonesianTextNormalizer.ts deleted file mode 100644 index b68315b..0000000 --- a/services/discord-gateway/src/modules/ai-moderation/indonesianTextNormalizer.ts +++ /dev/null @@ -1,430 +0,0 @@ -// No imports needed — pure rule-based, no external dependencies. - -const CUSTOM_EMOJI_PATTERN = //g; - -/** - * In-memory cache TTL (10 min) — avoids re-scanning identical text. - */ -const BADWORD_CACHE_TTL_MS = 10 * 60 * 1000; - -interface BadwordCacheEntry { - value: string[]; - expiresAt: number; -} - -const badwordCache = new Map(); - -// --------------------------------------------------------------------------- -// Safe pattern pre-filter — short-circuits definitively safe messages. -// Conservative: only returns true for patterns that CANNOT be violations. -// --------------------------------------------------------------------------- - -const SAFE_PATTERNS: Array<{ - test: (text: string) => boolean; - reason: string; -}> = [ - { - test: (t) => - /^(wkwk+|w+kw+k+|wkwkw+|haha+|hehe+|hihi+|huhu+|xixi+|wakak+|awkwa+)$/i.test( - t, - ), - reason: "laughter pattern", - }, - { - test: (t) => - /^(ok|oke|okay|sip|siap|aman|mantap|gas|gass|gaskeun|santuy|gaskan|lah|wih|wah|eh|nah|loh|hmm|hm|heh)$/i.test( - t, - ), - reason: "single-word affirmative", - }, - { - test: (t) => - /^(hai|halo|hello|hi|oi|woy|woi|pagi|siang|sore|malam|mlm|p|w|L|F|gws|thx|thks|makasih|ty|thanks|yw|sama-sama|ok sip|ok bang|siap bang)$/i.test( - t, - ), - reason: "greeting/common expression", - }, - { - test: (t) => t.length <= 2, - reason: "very short message (1-2 chars)", - }, - { - test: (t) => /^[\d\s.,!?;:'"()\-_]+$/.test(t), - reason: "numeric/punctuation only", - }, -]; - -// --------------------------------------------------------------------------- -// Rule-based Indonesian badword detection (NO LLM calls) -// -// Uses word-boundary regex matching to detect known Indonesian badwords. -// Context-aware: matches only whole words to avoid false positives like -// "asu" in "kasus", "kontol" in "rekontolasi". -// -// Each category maps to a flag the moderation LLM can use as context. -// --------------------------------------------------------------------------- - -interface BadwordEntry { - words: string[]; - flag: string; - description: string; -} - -const BADWORD_CATEGORIES: BadwordEntry[] = [ - { - description: "vulgar genitalia / sexual terms", - flag: "vulgar_language", - words: [ - "kontol", - "memek", - "pepek", - "tempik", - "peler", - "pelir", - "pukimak", - "pukima", - "jancok", - "jancuk", - "cok", - "cuk", - "pantek", - "palek", - "ngentot", - "ngewe", - "entot", - "ewe", - "coli", - "sange", - "sangean", - "ngocok", - "bangkot", - "nenen", - "tete", - "tetek", - "dodot", - "kentu", - "perek", - "bispak", - "bangsat", - "babi", - "asu", - "anjing", - "anjir", - "anjirt", - "njing", - "njir", - "anjay", - "kampret", - "kampang", - "brengsek", - "brengus", - "bejad", - "bajingan", - "goblok", - "tolol", - "bego", - "dungu", - "idiot", - "beban", - "keparat", - "setan", - "iblis", - "sialan", - "sial", - "kacang", - "edan", - "gila", - "titten", - "bitch", - "whore", - "slut" - ], - }, - { - description: "harassment / targeted insults", - flag: "harassment", - words: [ - "mampus", - "mati", - "bunuh", - "bacot", - "cupu", - "geblek", - "kere", - "ngawur", - "sembarangan", - "nyampah", - "nyampah", - "sarap", - "ke laut aja", - "gila lu", - "sinting", - "editan", - "mending mati", - "monyet", - "kuda", - "unta", - "bangke", - "bangsat", - "kys", - "kill yourself" - ], - }, - { - description: "SARA / racial slurs (non-exhaustive)", - flag: "sara", - words: [ - "cina", - "tionghoa", - "pribumi", - "non-pribumi", - "kaffir", - "kafir", - "murtad", - "sesat", - "liberal", - "komunis", - "komunisme", - "pki", - ], - }, - { - description: "gambling / judi", - flag: "gambling", - words: [ - "judi", - "slot", - "togel", - "toto gelap", - "casino", - "roulette", - "poker", - "domino", - "gaple", - "sabung ayam", - "bola jalan", - "maxwin", - "gacor", - "scatter", - "bonanza", - "olympus", - ], - }, - { - description: "hate speech / extreme discrimination", - flag: "hate_speech", - words: [ - "bencina", - "bencin", - "bangsat", - "dajjal", - "laknat", - "keparat", - "dasar cina", - "dasar tionghoa", - "dasar pribumi", - "nigger", - "nigga" - ], - }, -]; - -/** - * Build a single combined regex per category that matches whole words only. - * Uses word boundaries (\b) so "asu" matches "asu" but not "kasus". - * For multi-word entries, builds an alternation of the full phrases. - */ -const BADWORD_REGEX_CACHE = new Map(); - -function buildBadwordRegex(words: string[]): RegExp { - // Sort by length descending so longer phrases match before their substrings - const sorted = [...words].sort((a, b) => b.length - a.length); - // Escape regex special chars in each word - const escaped = sorted.map((w) => w.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")); - const pattern = escaped - .map((w) => { - // Multi-word phrases (containing space) — match as-is - if (w.includes("\\ ")) return w; - // Single word — word boundaries - return `\\b${w}\\b`; - }) - .join("|"); - return new RegExp(pattern, "i"); -} - -function matchBadwords(text: string): string[] { - const hits: Set = new Set(); - const lowerText = text.toLowerCase(); - - for (const category of BADWORD_CATEGORIES) { - let regex = BADWORD_REGEX_CACHE.get(category.flag); - if (!regex) { - regex = buildBadwordRegex(category.words); - BADWORD_REGEX_CACHE.set(category.flag, regex); - } - if (regex.test(lowerText)) { - hits.add(category.flag); - } - } - - return Array.from(hits); -} - -// --------------------------------------------------------------------------- -// Exported utilities -// --------------------------------------------------------------------------- - -/** - * Checks whether a text message is definitively safe and does not need - * badword detection at all. - */ -export function isDefinitivelySafe(text: string): boolean { - const { text: normalized } = normalizeDiscordCustomEmoji(text); - const trimmed = normalized.trim(); - if (trimmed.length === 0) return true; - return SAFE_PATTERNS.some((p) => p.test(trimmed)); -} - -export function normalizeDiscordCustomEmoji(text: string): { - text: string; - emojiNames: string[]; -} { - const emojiNames: string[] = []; - const normalized = text.replace( - CUSTOM_EMOJI_PATTERN, - (_match, name: string) => { - emojiNames.push(name); - return `[emoji:${name}]`; - }, - ); - return { text: normalized, emojiNames }; -} - -function normalizeBadwordCacheKey(text: string): string { - return text.trim().replace(/\s+/g, " ").toLowerCase(); -} - -function getCachedBadwords(key: string): string[] | null { - const entry = badwordCache.get(key); - if (!entry) return null; - if (entry.expiresAt <= Date.now()) { - badwordCache.delete(key); - return null; - } - return [...entry.value]; -} - -function setCachedBadwords(key: string, value: string[]): void { - badwordCache.set(key, { - value: [...new Set(value)], - expiresAt: Date.now() + BADWORD_CACHE_TTL_MS, - }); - - if (badwordCache.size > 500) { - const now = Date.now(); - for (const [cacheKey, entry] of badwordCache) { - if (entry.expiresAt <= now) { - badwordCache.delete(cacheKey); - } - } - if (badwordCache.size > 500) { - const oldestKeys = Array.from(badwordCache.entries()) - .sort((a, b) => a[1].expiresAt - b[1].expiresAt) - .slice(0, badwordCache.size - 500) - .map(([cacheKey]) => cacheKey); - for (const cacheKey of oldestKeys) { - badwordCache.delete(cacheKey); - } - } - } -} - -// --------------------------------------------------------------------------- -// PURE RULE-BASED badword detection (synchronous, no LLM calls) -// --------------------------------------------------------------------------- - -/** - * Detect badwords in text using pure rule-based matching. - * - * Previously used a 3-tier pipeline (in-memory → DB → LLM API call) that - * caused N+1 LLM calls per batch, multiplying costs by ~10x. - * - * Now uses word-boundary regex matching against known Indonesian badword - * categories. Fully synchronous — no DB, no API, no async overhead. - * - * Cache retained as a simple in-memory LRU for repeated identical texts. - */ -export function detectIndonesianBadwords(text: string): string[] { - const cacheKey = normalizeBadwordCacheKey(text); - - // ── In-memory cache (fastest) ── - const cached = getCachedBadwords(cacheKey); - if (cached) return cached; - - // ── Safe pre-filter ── - if (isDefinitivelySafe(text)) { - setCachedBadwords(cacheKey, []); - return []; - } - - // ── Rule-based matching ── - const hits = matchBadwords(text); - setCachedBadwords(cacheKey, hits); - return hits; -} - -// --------------------------------------------------------------------------- -// Synchronous evidence builders (no async needed anymore) -// --------------------------------------------------------------------------- - -export interface ModerationTextEvidence { - raw: string; - normalized: string; - notes: string[]; - badwords: string[]; - hasBadwords: boolean; -} - -export function buildModerationTextEvidence( - text: string, -): ModerationTextEvidence { - const emojiNormalized = normalizeDiscordCustomEmoji(text); - const badwordHits = detectIndonesianBadwords(emojiNormalized.text); - const notes: string[] = []; - - for (const emojiName of emojiNormalized.emojiNames) { - notes.push( - `emoji:${emojiName}=Discord custom emoji/expression; not text offense by default`, - ); - } - - if (badwordHits.length > 0) { - notes.push(`Known badword detected: ${badwordHits.join(", ")}`); - } else { - notes.push("no known badword detected"); - } - - return { - raw: text, - normalized: emojiNormalized.text, - notes: Array.from(new Set(notes)), - badwords: badwordHits, - hasBadwords: badwordHits.length > 0, - }; -} - -export function formatModerationTextEvidenceForPrompt(text: string): string { - const evidence = buildModerationTextEvidence(text); - if (evidence.normalized === evidence.raw && evidence.notes.length === 0) { - return ""; - } - - return [ - `[normalized_text: ${evidence.normalized}]`, - evidence.notes.length > 0 - ? `[normalization_notes: ${evidence.notes.join("; ")}]` - : null, - ] - .filter(Boolean) - .join(" "); -} diff --git a/services/discord-gateway/src/modules/ai-moderation/llmModerationClient.ts b/services/discord-gateway/src/modules/ai-moderation/llmModerationClient.ts index 6a94c1d..f8973f7 100644 --- a/services/discord-gateway/src/modules/ai-moderation/llmModerationClient.ts +++ b/services/discord-gateway/src/modules/ai-moderation/llmModerationClient.ts @@ -11,7 +11,7 @@ import type { AttachmentRecord, MessageRecord, } from "../message-capture/types.js"; -import { formatModerationTextEvidenceForPrompt } from "./indonesianTextNormalizer.js"; + import { llmChat, llmVision } from "./llmClient.js"; import { buildSystemPrompt as buildSystemPromptModular } from "./moderationPrompt.js"; import { @@ -999,16 +999,6 @@ async function runTextOnlyBatch( const maxBatchSize = config.AI_LLM_TEXT_BATCH_SIZE ?? 20; const timeoutMs = config.AI_LLM_MEDIA_ANALYSIS_TIMEOUT_MS ?? 60000; - // Pre-compute text evidence (normalization + badword detection) - const textEvidenceMap = new Map(); - await Promise.all( - targets.map(async (msg) => { - const content = msg.edited_content ?? msg.content; - const evidence = formatModerationTextEvidenceForPrompt(content); - textEvidenceMap.set(msg.id, evidence); - }), - ); - // ── Fetch web content from URLs in text-only messages ── // Prevents LLM from guessing based on domain name alone (e.g., false "scam" flags). // The fetched text content is injected into the message XML so the LLM can @@ -1156,8 +1146,6 @@ async function runTextOnlyBatch( const messagesBlock = batch .map((msg) => { const content = getAnalysisContent(msg); - const textEvidence = textEvidenceMap.get(msg.id) ?? ""; - const textContext = textEvidence ? `\n${textEvidence}` : ""; // Inject fetched web content for URLs found in this message const msgUrls = extractUrlsFromText(content); @@ -1173,7 +1161,7 @@ async function runTextOnlyBatch( const userCtx = userContexts.get(msg.user_id) ?? ""; // XML delimiters wrap each message for prompt safety (R1) - return `\n ${userCtx}\n ${content}${textContext}${webContext}\n`; + return `\n ${userCtx}\n ${content}${webContext}\n`; }) .join("\n"); @@ -1652,12 +1640,9 @@ async function _runSingleMediaAnalysis( ); // ── 5. Build single-message prompt with XML delimiters (R1) ── - const textEvidence = formatModerationTextEvidenceForPrompt(content); - const webTexts = webTextMap.get(targetId) ?? []; const mediaAnalyses = mediaAnalysisMap.get(targetId) ?? []; const webContext = webTexts.length > 0 ? `\n${webTexts.join("\n")}` : ""; - const textContext = textEvidence ? `\n${textEvidence}` : ""; const mediaAnalysisContext = mediaAnalyses.length > 0 ? `\n${mediaAnalyses.join("\n")}` : ""; @@ -1693,7 +1678,7 @@ async function _runSingleMediaAnalysis( const userCtx = `${historyStr}`; // XML delimiters wrap the message content (R1) - const messageBlock = `\n ${userCtx}\n ${content}${mediaContext ? ` ${mediaContext}` : ""}${textContext}${webContext}${mediaAnalysisContext}\n`; + const messageBlock = `\n ${userCtx}\n ${content}${mediaContext ? ` ${mediaContext}` : ""}${webContext}${mediaAnalysisContext}\n`; // Modular system prompt with XML delimiters (R1, R7, R8) const correctedExamples = await buildCorrectedFewShotExamples();