diff --git a/services/discord-gateway/src/modules/ai-moderation/ai-analysis-worker.ts b/services/discord-gateway/src/modules/ai-moderation/ai-analysis-worker.ts index 0f3c221..08726f8 100644 --- a/services/discord-gateway/src/modules/ai-moderation/ai-analysis-worker.ts +++ b/services/discord-gateway/src/modules/ai-moderation/ai-analysis-worker.ts @@ -1,20 +1,15 @@ /** * ai-analysis-worker.ts * - * Two-pass AI moderation analysis worker (Piscina-compatible). + * AI moderation analysis worker (Piscina-compatible). * * ## Pipeline * - * Message → Layer 1 (fast classifier / heuristic) - * │ - * ├─ clear match → final result (no LLM call) - * └─ ambiguous → Layer 2 (LLM evaluator) + * Message → LLM evaluator (with conversation context + media evidence) * - * Layer 1 runs synchronously in-memory. Layer 2 calls the LLM API via - * the existing moderation pipeline (moderationOrchestrator). - * - * This file replaces the old `aiAnalysisWorker.ts` (archived) with a - * simpler, unified worker that combines both layers. + * Every message is judged by the LLM — there is no regex/heuristic + * pre-classification. A failed LLM call yields an explicit "error" status + * (never a heuristic verdict), and the recovery worker retries it later. */ import { createChildLogger } from "@/shared/logger/index"; @@ -23,8 +18,6 @@ import { initializeDatabase } from "../../shared/database/drizzle.js"; import { messageStore } from "../message-capture/messageStore.js"; import type { MessageRecord } from "../message-capture/types.js"; import { buildConversationContext } from "./conversationContext.js"; -import { classifyMessage } from "./fastClassifier.js"; -import type { Layer1Result } from "./fastClassifier.js"; import { runModerationAnalysis } from "./moderationOrchestrator.js"; const logger = createChildLogger("ai-analysis-worker"); @@ -89,11 +82,7 @@ export interface MessageBatch { // Worker job types (Piscina entry point) type WorkerJob = | { type: "batch"; conversationKey: string; messages: MessageRecord[] } - | { - type: "individual"; - message: MessageRecord; - skipNormalAnalysis: boolean; - }; + | { type: "individual"; message: MessageRecord; skipNormalAnalysis: boolean }; type BatchOkResponse = { ok: true; @@ -186,39 +175,21 @@ export default async function workerRouter( } // --------------------------------------------------------------------------- -// Two-pass pipeline +// Single-pass LLM pipeline // --------------------------------------------------------------------------- /** - * Runs the two-pass pipeline on a single message: - * 1. Layer 1 — fast heuristic classifier - * 2. Layer 2 (if cascade) — LLM-based evaluator + * Runs the LLM moderation analysis on a single message. * - * Returns the combined AnalysisResult. + * The LLM verdict IS the result — confidence, severity, flags and + * explanation all come from the model. On failure the message is marked + * "error" (explicit, retryable) instead of receiving a heuristic verdict. */ -async function runTwoPassPipeline( +async function runLLMAnalysis( message: MessageRecord, contextText: string, attachments: Awaited>, ): Promise { - // ── Layer 1: Fast classifier ────────────────────────────────────────── - const layer1Result: Layer1Result = classifyMessage(message); - - logger.debug( - { - messageId: message.id, - layer1Flags: layer1Result.flags, - cascade: layer1Result.cascadeToLayer2, - }, - "Layer 1 classification complete", - ); - - if (!layer1Result.cascadeToLayer2) { - // Layer 1 result is final — no LLM call needed - return buildResultFromLayer1(message.id, layer1Result); - } - - // ── Layer 2: LLM-based evaluation ───────────────────────────────────── try { const moderationResult = await runModerationAnalysis({ targets: [message], @@ -231,45 +202,44 @@ async function runTwoPassPipeline( } const llmResult = moderationResult.results[0] as unknown as AnalysisResult; - return mergeLayers(layer1Result, llmResult); + if (llmResult.status === "error") { + return llmResult; + } + + return { + messageId: llmResult.messageId, + status: llmResult.status ?? "clean", + flags: llmResult.flags ?? [], + categories: llmResult.categories ?? [], + severity: llmResult.severity ?? "none", + confidence: normalizeConfidence(llmResult.confidence), + recommendedAction: llmResult.recommendedAction ?? "none", + toxicityScore: llmResult.toxicityScore ?? 0, + harmScore: llmResult.harmScore ?? 0, + jailbreakScore: llmResult.jailbreakScore ?? 0, + safetyScore: llmResult.safetyScore ?? 0, + explanation: + llmResult.explanation?.trim() || + (llmResult.status === "clean" + ? "Tidak ada indikasi pelanggaran." + : "Pesan terindikasi melanggar kebijakan (analisis AI)."), + }; } catch (error) { const errorMsg = error instanceof Error ? error.message : String(error); logger.warn( { messageId: message.id, error: errorMsg }, - "Layer 2 (LLM) analysis failed — falling back to Layer 1 result", + "LLM analysis failed for message", ); - // Fallback to Layer 1 with reduced confidence - const fallback = buildResultFromLayer1(message.id, layer1Result); - fallback.confidence = Math.min(fallback.confidence, 0.4); - return fallback; + return buildFallbackResult(message.id, errorMsg); } } -// --------------------------------------------------------------------------- -// Result builders -// --------------------------------------------------------------------------- - -function buildResultFromLayer1( - messageId: string, - layer1: Layer1Result, -): AnalysisResult { - const status = layer1.severity === "none" ? "clean" as const : "flagged" as const; - const recommendedAction = mapSeverityToAction(layer1.severity); - - return { - messageId, - status, - flags: layer1.flags, - categories: layer1.flags, - severity: layer1.severity === "high" ? "high" as const : layer1.severity === "medium" ? "medium" as const : "low" as const, - confidence: layer1.confidence, - recommendedAction, - toxicityScore: layer1.toxicityScore, - harmScore: layer1.harmScore, - jailbreakScore: 0, - safetyScore: 0, - explanation: layer1.explanation, - }; +/** Clamp LLM-provided confidence to [0, 1]; default by status when missing. */ +function normalizeConfidence(raw: number | undefined | null): number { + if (typeof raw === "number" && Number.isFinite(raw)) { + return Math.min(Math.max(raw, 0), 1); + } + return 0.7; } function buildFallbackResult( @@ -292,70 +262,6 @@ function buildFallbackResult( }; } -function mergeLayers( - layer1: Layer1Result, - llmResult: AnalysisResult, -): AnalysisResult { - // Merge flags from both layers (deduplicate) - const flagSet = new Set([...layer1.flags, ...(llmResult.flags || [])]); - - // Take the max severity - const severityOrder = ["none", "low", "medium", "high", "critical"] as const; - const l1Idx = severityOrder.indexOf(layer1.severity); - const l2Idx = severityOrder.indexOf( - (llmResult.severity ?? "none") as (typeof severityOrder)[number], - ); - const finalSeverity = severityOrder[Math.max(l1Idx, l2Idx)]; - - // Combined confidence: weighted average favoring LLM when available - const combinedConfidence = - 0.3 * layer1.confidence + 0.7 * (llmResult.confidence ?? 0.5); - - // Combine scores (take max per dimension) - const toxicityScore = Math.max( - layer1.toxicityScore, - llmResult.toxicityScore ?? 0, - ); - const harmScore = Math.max(layer1.harmScore, llmResult.harmScore ?? 0); - - return { - messageId: llmResult.messageId, - status: llmResult.status === "error" ? "error" as const : llmResult.status ?? "clean" as const, - flags: Array.from(flagSet), - categories: [ - ...new Set([ - ...layer1.flags, - ...(llmResult.categories ?? []), - ]), - ], - severity: finalSeverity, - confidence: Math.min(combinedConfidence, 1), - recommendedAction: llmResult.recommendedAction ?? mapSeverityToAction(finalSeverity), - toxicityScore, - harmScore, - jailbreakScore: llmResult.jailbreakScore ?? 0, - safetyScore: llmResult.safetyScore ?? 0, - explanation: llmResult.explanation ?? layer1.explanation, - }; -} - -function mapSeverityToAction( - severity: "none" | "low" | "medium" | "high" | "critical", -): AnalysisResult["recommendedAction"] { - switch (severity) { - case "none": - return "none"; - case "low": - return "monitor"; - case "medium": - return "review"; - case "high": - return "delete"; - case "critical": - return "escalate"; - } -} - // --------------------------------------------------------------------------- // Batch handler // --------------------------------------------------------------------------- @@ -391,16 +297,16 @@ async function processBatch(job: { const attachments = await messageStore.getAttachmentsForMessages(allMessageIds); - // Run two-pass pipeline for each message + // Run LLM analysis for each message const analysisResults = await Promise.all( messages.map(async (msg) => { try { - return await runTwoPassPipeline(msg, contextText, attachments); + return await runLLMAnalysis(msg, contextText, attachments); } catch (error) { const errorMsg = error instanceof Error ? error.message : String(error); logger.error( { messageId: msg.id, error: errorMsg }, - "Two-pass pipeline failed for message", + "LLM analysis failed for message", ); return buildFallbackResult(msg.id, errorMsg); } @@ -435,7 +341,7 @@ async function processBatch(job: { saved: allRows.length, conversationKey, }, - "Two-pass batch analysis complete", + "LLM batch analysis complete", ); return { ok: true, conversationKey, rows: allRows }; @@ -450,22 +356,11 @@ async function processIndividual(job: { message: MessageRecord; skipNormalAnalysis: boolean; }): Promise { - const { message, skipNormalAnalysis } = job; + const { message } = job; - if (skipNormalAnalysis) { - // Use Layer 1 only (fast) - const layer1Result = classifyMessage(message); - - if (!layer1Result.cascadeToLayer2) { - // Layer 1 is sufficient - return { - ok: true, - results: [buildResultFromLayer1(message.id, layer1Result)], - }; - } - } - - // Full analysis (context + two-pass) + // Full analysis (context + LLM). `skipNormalAnalysis` is accepted for + // compatibility with the old two-pass protocol but always runs the LLM — + // there is no heuristic path anymore. const contextBefore = await messageStore.getConversationContextBefore({ channelId: message.channel_id, threadId: message.thread_id, @@ -487,14 +382,14 @@ async function processIndividual(job: { ]); try { - const result = await runTwoPassPipeline(message, contextText, attachments); + const result = await runLLMAnalysis(message, contextText, attachments); return { ok: true, results: [result] }; } catch (error) { const errorMsg = error instanceof Error ? error.message : String(error); logger.error( { messageId: message.id, error: errorMsg }, - "Individual two-pass analysis failed", + "Individual LLM analysis failed", ); return { ok: true, results: [buildFallbackResult(message.id, errorMsg)] }; } -} \ No newline at end of file +} diff --git a/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts b/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts index 93e45ca..8f831c9 100644 --- a/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts +++ b/services/discord-gateway/src/modules/ai-moderation/conversationContext.ts @@ -2,6 +2,7 @@ import { createChildLogger } from "@/shared/logger/index"; import { encoding_for_model as encodingForModel } from "tiktoken"; import { formatMediaEvidenceForPrompt } from "../message-capture/messageMetadata.js"; import type { MessageRecord } from "../message-capture/types.js"; +import { sanitizeDiscordTokens } from "./discordTokens.js"; const logger = createChildLogger("conversationContext"); @@ -67,7 +68,9 @@ export function formatMessageForPrompt( msg: MessageRecord, label: "context" | "target", ): string { - const content = msg.edited_content ?? msg.content; + const content = sanitizeDiscordTokens( + msg.edited_content ?? msg.content, + ); const timestamp = formatTimestamp(msg.created_at); const mediaEvidence = formatMediaEvidenceForPrompt(msg.metadata); const mediaSuffix = mediaEvidence ? ` ${mediaEvidence}` : ""; diff --git a/services/discord-gateway/src/modules/ai-moderation/discordTokens.ts b/services/discord-gateway/src/modules/ai-moderation/discordTokens.ts new file mode 100644 index 0000000..d7e73fe --- /dev/null +++ b/services/discord-gateway/src/modules/ai-moderation/discordTokens.ts @@ -0,0 +1,41 @@ +/** + * discordTokens.ts + * + * Normalizes Discord markdown tokens (custom emoji, user/role/channel + * mentions, timestamps) into readable placeholders before content reaches + * the LLM. The numeric snowflake IDs inside these tokens are meaningless to + * an LLM and were the source of repeated false positives in the old + * regex-based Layer 1 classifier — so raw IDs are never presented. + */ + +// <:name:id> | | <@id> | <@!id> | <@&id> | <#id> | +const DISCORD_TOKEN_RE = + /<(?:a?:([^>]{1,32}):(\d{17,20})|@!?(\d{17,20})|@&(\d{17,20})|#(\d{17,20})|t:(\d{10,11})(?::([tTdDRfF]))?)>/g; + +/** + * Replaces Discord markdown tokens with readable placeholders. + * + * - `<:name:id>` / `` → `[emoji:name]` + * - `<@123>` / `<@!123>` → `@user` + * - `<@&123>` → `@role` + * - `<#123>` → `#channel` + * - `` → `[time]` + * + * No numeric IDs survive, so digit-shaped patterns can never match them. + */ +export function sanitizeDiscordTokens(content: string): string { + if (!content.includes("<")) return content; + return content.replace( + DISCORD_TOKEN_RE, + (_full, emojiName, _emojiId, _userId, _roleId, _channelId, _time, _style) => { + if (emojiName) return `[emoji:${emojiName}]`; + // Capture groups tell us which alternative matched by position: + // 3=user, 4=role, 5=channel, 6=time + if (_userId !== undefined) return "@user"; + if (_roleId !== undefined) return "@role"; + if (_channelId !== undefined) return "#channel"; + if (_time !== undefined) return "[time]"; + return " "; + }, + ); +} diff --git a/services/discord-gateway/src/modules/ai-moderation/fastClassifier.ts b/services/discord-gateway/src/modules/ai-moderation/fastClassifier.ts deleted file mode 100644 index c707387..0000000 --- a/services/discord-gateway/src/modules/ai-moderation/fastClassifier.ts +++ /dev/null @@ -1,419 +0,0 @@ -/** - * fastClassifier.ts - * - * Layer 1 — Synchronous heuristic classifier for the two-pass moderation pipeline. - * Runs BEFORE any LLM call. Catches obvious spam, NSFW patterns, repeated characters, - * and other low-hanging fruit with zero network cost. - * - * When a strong heuristic match is found, `cascadeToLayer2` is `false` and the - * result is used as the final verdict. Otherwise the message proceeds to the LLM. - */ - -import type { MessageRecord } from "../message-capture/types.js"; - -// --------------------------------------------------------------------------- -// Types -// --------------------------------------------------------------------------- - -export interface Layer1Result { - flags: string[]; - severity: "none" | "low" | "medium" | "high"; - toxicityScore: number; - harmScore: number; - cascadeToLayer2: boolean; - confidence: number; - explanation: string; -} - -// --------------------------------------------------------------------------- -// Pattern definitions -// --------------------------------------------------------------------------- - -interface Pattern { - name: string; - test: (content: string, mentions: number) => boolean; - severity: "low" | "medium" | "high"; - score: number; // contribution to toxicity/harm score - category: "toxicity" | "harm" | "spam" | "safety"; -} - -// ── Zalgo / zero-width detection ───────────────────────────────────────── - -const ZALGO_RE = - new RegExp( - '[̀-ͯ҃-҉ؐ-ًؚ-ٰٟۖ-ۜ۟-ۤۧ-۪ۨ-ܑۭܰ-݊ަ-ްँ-ः़ा-्॑-॔ॢ-ॣঁ-ঃ়া-ৄে-ৈো-্ৗৢ-ৣ৾ਁ-ਃ਼ਾ-ੂੇ-ੈੋ-੍ੑੰ-ੱੵઁ-ઃ઼ા-ૅે-ૉો-્ૢ-ૣૺ-૿ଁ-ଃ଼ା-ୄେ-ୈୋ-୍ୖ-ୗୢ-ୣஂா-ூெ-ைொ-்ௗఀ-ఃా-ౄె-ైొ-్ౕ-ౖౢ-ౣಁ-ಃ಼ಾ-ೄೆ-ೈೊ-್ೕ-ೖೢ-ೣഀ-ഃ഻-഼ാ-ൄെ-ൈൊ-്ൗൢ-ൣඁ-ඃ්ා-ුූෘ-ෟෲ-ෳัิ-ฺ็-๎ັິ-ູົ-ຼ່-ໍ༘-༹༙༵༷༾-༿ཱ-྄྆-྇ྍ-ྗྙ-ྼ࿆ါ-ှၖ-ၙၞ-ၠၢ-ၤၧ-ၭၱ-ၴႂ-ႍႏႚ-ႝ፝-፟ᜒ-᜔ᜲ-᜴ᝒ-ᝓᝲ-ᝳ឴-៓៝᠋-᠍ᢩᤠ-ᤫᤰ-᤻ᨗ-ᨛᩕ-ᩞ᩠-᩿᩼᪰-᪾ᬀ-ᬄ᬴-᭄᭫-᭳ᮀ-ᮁᮢ-ᮥᮨ-ᮩ᮫-ᮭ᯦-᯳ᰤ-᰷᳐-᳔᳒-᳨᳭ᳲ-᳴᳷-᳹᷀-᷿​-‏
-  -⃐-⃰⳯-⵿⳱ⷠ-〪ⷿ-゙〯-゚꙯-꙲ꙴ-꙽ꚞ-ꚟ꛰-꛱ꠂ꠆ꠋꠣ-ꠧ꠬ꢀ-ꢁꢴ-ꣅ꣠-꣱ꣿ-꤉ꤦ-꤭ꥇ-꥓ꥠ-ꥼꦀ-ꦃ꦳-꧀ꧥꨩ-ꨶꩃꩌꩍꩻ-ꩽꪰꪲ-ꪴꪷ-ꪸꪾ-꪿꫁ꫫ-ꫯꫵ-꫶ꯣ-ꯪ꯬꯭ﬞ︀-️︠-︯-]|­|͏|؜ᅟᅠ឴឵᠎ -   ⁠-⁤⁦-ㅤᅠ￰-￸\u{1e001}\u{1e020}-\u{1e07f}', -); - -const ZERO_WIDTH_RE = /[​-‍⁠­؜]/; - -// ── URL / invite / phone / email / crypto patterns ───────────────────── - -const URL_RE = /https?:\/\/[^\s"]+/gi; -const INVITE_RE = /(?:discord\.(?:gg|com\/invite)|dsc\.gg)\/[a-zA-Z0-9_-]+/gi; -const INVITE_CODE_RE = /(?:^|\s)([a-zA-Z0-9_-]{6,12})(?:\s|$)/g; -// Word-boundary phone match. `(?, ), user/role/channel mentions and -// timestamps embed long numeric snowflakes. If left in the text, digit-based -// patterns (phone_number, personal_info, ip_address_sharing) false-positive -// on them — e.g. <:mambotongue:1463255254220148939> "matched" phone_number. -const DISCORD_TOKEN_RE = - /<(?:a?:[^>]{1,32}:\d{17,20}|@!?\d{17,20}|@&\d{17,20}|#\d{17,20}|t:\d{10,11}(?::[tTdDRfF])?)>/g; - -/** Replaces Discord markdown tokens with a space so they can't trip patterns. */ -export function sanitizeDiscordTokens(content: string): string { - return content.replace(DISCORD_TOKEN_RE, " "); -} - -// ── Spam / low-quality patterns ───────────────────────────────────────── - -const REPEATED_CHAR_RE = /(.)\1{8,}/; // 9+ repeated chars -const REPEATED_WORD_RE = /\b(\w{3,})\b\s*\b\1\b\s*\b\1\b/; // same word 3x -const EXCESSIVE_CAPS_RE = /[A-Z]{6,}/; -const EXCESSIVE_EMOJI_RE = - /[\u{1F300}-\u{1F9FF}\u{2600}-\u{26FF}\u{2700}-\u{27BF}]/gu; -const BASE64_RE = - /^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/; -const PHISHING_RE = - /(?:free\s*(?:nitro|gift|prime|steam|vbucks?)|click\s*(?:here|this)\s*(?:to|for)\s*(?:claim|win|verify)|login\s*:\s*\w+\s*password\s*:\s*\w+)/gi; - -// ── Toxicity patterns ─────────────────────────────────────────────────── - -const HARASSMENT_RE = - /\b(?:fuck|shit|asshole|bitch|dickhead|cunt|motherfucker|bastard|piss\s*off|screw\s*you|go\s*(?:to\s*)?hell|kys|kill\s*(?:yourself| urself))\b/i; -const HATE_SPEECH_RE = - /\b(?:nazi|white\s*supremacy|heil|racial\s*purity|race\s*war)\b/i; - -// ── Harm patterns ─────────────────────────────────────────────────────── - -const SELF_HARM_RE = - /\b(?:kill\s*(?:myself|me)|end\s*(?:my|the)\s*(?:life|own)|suicide|want\s*(?:to\s*)?die|cut\s*(?:myself|my\s*wrists)|harm\s*myself)\b/i; -const VIOLENCE_RE = - /\b(?:shoot|stab|bomb|massacre|terrorist|behead|torture|murder)\b/i; - -// ── Safety patterns ───────────────────────────────────────────────────── - -const PERSONAL_INFO_RE = - /\b(?:\d{3}-\d{2}-\d{4}|(?:\d{3}\s?){2}\d{4})\b/; // SSN / IDs -const SEXTORTION_RE = - /\b(?:nudes?\s*(?:pic|photo|video|send|trade)|cp\s*(?:content|link|loli|shotacon)|underage|minor\s*(?:girl|boy|content))\b/i; -const GROOMING_RE = - /\b(?:how\s*old\s*are\s*you|are\s*you\s*(?:alone|home\s*alone)|dm\s*me\s*(?:baby|honey|sweetie|cutie)|send\s*(?:nudes|pics))\b/i; - -// ── Mass-mention patterns ─────────────────────────────────────────────── - -const EVERYONE_MENTION = /@everyone/g; -const HERE_MENTION = /@here/g; -const ROLE_MENTION = /<@&(\d+)>/g; - -// ── Pattern registry (ordered roughly by specificity) ─────────────────── - -const PATTERNS: Pattern[] = [ - // ── High severity ───────────────────────────────────────────────── - { - name: "self_harm", - test: (c) => SELF_HARM_RE.test(c), - severity: "high", - score: 0.9, - category: "harm", - }, - { - name: "violence_threat", - test: (c) => VIOLENCE_RE.test(c), - severity: "high", - score: 0.85, - category: "harm", - }, - { - name: "sextortion", - test: (c) => SEXTORTION_RE.test(c), - severity: "high", - score: 0.95, - category: "safety", - }, - { - name: "grooming", - test: (c) => GROOMING_RE.test(c), - severity: "high", - score: 0.9, - category: "safety", - }, - { - name: "hate_speech", - test: (c) => HATE_SPEECH_RE.test(c), - severity: "high", - score: 0.85, - category: "toxicity", - }, - { - name: "harassment", - test: (c) => HARASSMENT_RE.test(c), - severity: "medium", - score: 0.6, - category: "toxicity", - }, - { - name: "phishing", - test: (c) => PHISHING_RE.test(c), - severity: "high", - score: 0.85, - category: "harm", - }, - // ── Spam / low quality ──────────────────────────────────────────── - { - name: "mass_everyone_mention", - test: (_c, mentions) => mentions >= 3, - severity: "medium", - score: 0.65, - category: "spam", - }, - { - name: "excessive_caps", - test: (c) => { - const caps = (c.match(EXCESSIVE_CAPS_RE) || []).join(""); - return caps.length > 0 && caps.length / Math.max(c.length, 1) > 0.5; - }, - severity: "low", - score: 0.3, - category: "spam", - }, - { - name: "repeated_characters", - test: (c) => REPEATED_CHAR_RE.test(c), - severity: "low", - score: 0.25, - category: "spam", - }, - { - name: "repeated_words", - test: (c) => REPEATED_WORD_RE.test(c), - severity: "low", - score: 0.25, - category: "spam", - }, - { - name: "zalgo_text", - test: (c) => ZALGO_RE.test(c) || ZERO_WIDTH_RE.test(c), - severity: "medium", - score: 0.6, - category: "spam", - }, - { - name: "excessive_emojis", - test: (c) => { - const emojiCount = (c.match(EXCESSIVE_EMOJI_RE) || []).length; - const textLen = c.replace(EXCESSIVE_EMOJI_RE, "").trim().length; - return emojiCount >= 5 && textLen < emojiCount; - }, - severity: "low", - score: 0.2, - category: "spam", - }, - { - name: "base64_gibberish", - test: (c) => c.length >= 20 && BASE64_RE.test(c.trim()), - severity: "low", - score: 0.2, - category: "spam", - }, - // ── Medium severity ─────────────────────────────────────────────── - { - name: "personal_info", - test: (c) => PERSONAL_INFO_RE.test(c), - severity: "medium", - score: 0.6, - category: "safety", - }, - { - name: "discord_invite", - test: (c) => INVITE_RE.test(c), - severity: "low", - score: 0.2, - category: "spam", - }, - { - name: "url_only", - test: (c) => { - const urls = c.match(URL_RE); - if (!urls) return false; - const textWithoutUrls = c.replace(URL_RE, "").trim(); - return urls.length >= 3 && textWithoutUrls.length === 0; - }, - severity: "low", - score: 0.25, - category: "spam", - }, - { - name: "phone_number", - test: (c) => PHONE_RE.test(c), - severity: "medium", - score: 0.5, - category: "safety", - }, - { - name: "email_address", - test: (c) => EMAIL_RE.test(c), - severity: "low", - score: 0.3, - category: "safety", - }, - { - name: "crypto_address", - test: (c) => CRYPTO_RE.test(c), - severity: "medium", - score: 0.5, - category: "spam", - }, - { - name: "ip_address_sharing", - test: (c) => IP_RE.test(c), - severity: "low", - score: 0.3, - category: "safety", - }, -]; - -// --------------------------------------------------------------------------- -// Pattern matcher — count mentions -// --------------------------------------------------------------------------- - -function countMentions(content: string): number { - let count = 0; - const everyoneMatches = content.match(EVERYONE_MENTION); - if (everyoneMatches) count += everyoneMatches.length; - const hereMatches = content.match(HERE_MENTION); - if (hereMatches) count += hereMatches.length; - const roleMatches = content.match(ROLE_MENTION); - if (roleMatches) count += roleMatches.length; - return count; -} - -// --------------------------------------------------------------------------- -// Main classifier -// --------------------------------------------------------------------------- - -/** - * Runs Layer 1 heuristic classification on a message. - * Returns a `Layer1Result` with matched flags and a decision on - * whether to cascade to Layer 2 (LLM). - */ -export function classifyMessage(message: MessageRecord): Layer1Result { - const content = message.edited_content ?? message.content; - if (!content || content.trim().length === 0) { - return { - flags: [], - severity: "none", - toxicityScore: 0, - harmScore: 0, - cascadeToLayer2: true, // empty content still needs metadata check - confidence: 0, - explanation: "No text content to classify", - }; - } - - const mentions = countMentions(content); - // Strip Discord markdown tokens (emoji/mention snowflakes) before pattern - // matching so digit-based patterns don't false-positive on them. - const sanitized = sanitizeDiscordTokens(content); - const matchedFlags: string[] = []; - const severityWeights: Record = { - low: 1, - medium: 2, - high: 3, - }; - - let maxSeverityWeight = 0; - let totalToxicityScore = 0; - let totalHarmScore = 0; - let totalSafetyScore = 0; - let totalSpamScore = 0; - const matchedPatternDetails: string[] = []; - - for (const pattern of PATTERNS) { - if (pattern.test(sanitized, mentions)) { - matchedFlags.push(pattern.name); - matchedPatternDetails.push(pattern.name); - - const weight = severityWeights[pattern.severity] || 1; - maxSeverityWeight = Math.max(maxSeverityWeight, weight); - - switch (pattern.category) { - case "toxicity": - totalToxicityScore += pattern.score; - break; - case "harm": - totalHarmScore += pattern.score; - break; - case "safety": - totalSafetyScore += pattern.score; - break; - case "spam": - totalSpamScore += pattern.score; - break; - } - } - } - - // ── Determine final severity ─────────────────────────────────────── - - let finalSeverity: "none" | "low" | "medium" | "high" = "none"; - if (maxSeverityWeight >= 3) finalSeverity = "high"; - else if (maxSeverityWeight >= 2) finalSeverity = "medium"; - else if (maxSeverityWeight >= 1) finalSeverity = "low"; - - // ── Determine if we should cascade ───────────────────────────────── - // Cascade to Layer 2 whenever: - // 1. No high-severity match was found, OR - // 2. Only spam/low-quality patterns matched (need LLM for nuance) - // Do NOT cascade when a clear high-severity harm/safety/toxicity match - // was found — the heuristic is sufficient. - - const hasHighSeverityPattern = matchedFlags.some((f) => { - const p = PATTERNS.find((p) => p.name === f); - return p && p.severity === "high"; - }); - const hasOnlyLowSeveritySpam = matchedFlags.every((f) => { - const p = PATTERNS.find((p) => p.name === f); - return p && p.category === "spam" && p.severity !== "high"; - }); - - // Cascade if no matches, only spam, or low/medium severity toxicity/safety - const cascadeToLayer2 = - matchedFlags.length === 0 || - hasOnlyLowSeveritySpam || - (finalSeverity !== "high" && hasHighSeverityPattern === false); - - // ── Compute final scores (clamped 0-1) ───────────────────────────── - - const toxicityScore = Math.min(totalToxicityScore, 1); - const harmScore = Math.min(totalHarmScore, 1); - - // ── Confidence ───────────────────────────────────────────────────── - - const confidence = cascadeToLayer2 - ? 0.4 + 0.1 * matchedFlags.length // low confidence when punting to LLM - : 0.6 + 0.4 * (1 - matchedFlags.length / PATTERNS.length); - - const explanation = - matchedFlags.length > 0 - ? `Layer 1 matched: ${matchedPatternDetails.join(", ")}` - : "No heuristic patterns matched"; - - return { - flags: matchedFlags, - severity: finalSeverity, - toxicityScore, - harmScore, - cascadeToLayer2, - confidence: Math.min(confidence, 0.99), - explanation, - }; -} \ No newline at end of file diff --git a/services/discord-gateway/src/modules/ai-moderation/index.ts b/services/discord-gateway/src/modules/ai-moderation/index.ts index 026e745..996610d 100644 --- a/services/discord-gateway/src/modules/ai-moderation/index.ts +++ b/services/discord-gateway/src/modules/ai-moderation/index.ts @@ -1,9 +1,12 @@ export { startPendingAIAnalysisWorker } from "./aiAnalyzer.js"; export { runModerationAnalysis } from "./moderationOrchestrator.js"; export { buildSystemPrompt } from "./moderationPrompt.js"; -export { runSimpleTextFallback } from "./simpleFallback.js"; +export { sanitizeDiscordTokens } from "./discordTokens.js"; -// ── New two-pass pipeline exports ────────────────────────────────────────── -export { classifyMessage } from "./fastClassifier.js"; -export type { Layer1Result } from "./fastClassifier.js"; -export type { AnalysisInput, AnalysisResult, WorkerConfig, MessageBatch } from "./ai-analysis-worker.js"; \ No newline at end of file +// ── Single-pass LLM pipeline exports ───────────────────────────────────── +export type { + AnalysisInput, + AnalysisResult, + WorkerConfig, + MessageBatch, +} from "./ai-analysis-worker.js"; diff --git a/services/discord-gateway/src/modules/ai-moderation/individualFallbackProcessor.ts b/services/discord-gateway/src/modules/ai-moderation/individualFallbackProcessor.ts index aa271e4..abd81a8 100644 --- a/services/discord-gateway/src/modules/ai-moderation/individualFallbackProcessor.ts +++ b/services/discord-gateway/src/modules/ai-moderation/individualFallbackProcessor.ts @@ -82,7 +82,6 @@ async function processIndividualFallback( | { ok: false; results: AnalysisResult[]; error: string }; let analysisResult: { results: AnalysisResult[] } | null = null; - let usedSimpleFallback = false; if (workerResult.ok) { const stillIncomplete = workerResult.results.some((r) => @@ -96,38 +95,12 @@ async function processIndividualFallback( } } - // Step 2: If normal analysis failed, try SIMPLE fallback via worker - if (!analysisResult) { - logger.info( - { messageId }, - "Normal analysis failed -- trying simple text fallback via worker", - ); - - const simpleResult = (await workerPool.run({ - type: "individual", - message, - skipNormalAnalysis: true, - } as unknown)) as - | { ok: true; results: AnalysisResult[] } - | { ok: false; results: AnalysisResult[]; error: string }; - - if (simpleResult.ok) { - analysisResult = simpleResult; - usedSimpleFallback = true; - exhaustedOnIncomplete = false; - } - } - + // No heuristic fallback: an incomplete/errored LLM result stays a + // retryable error — the recovery worker picks it up later. Producing a + // regex/wordlist verdict here would reintroduce false positives. if (!analysisResult) { throw new Error( - `Both normal and simple analysis failed for message ${messageId}`, - ); - } - - if (usedSimpleFallback) { - logger.info( - { messageId, status: analysisResult.results[0]?.status }, - "Used simple text fallback for individual message (via worker)", + `LLM analysis failed for message ${messageId}`, ); } diff --git a/services/discord-gateway/src/modules/ai-moderation/moderationBuilders.ts b/services/discord-gateway/src/modules/ai-moderation/moderationBuilders.ts index 3c304d0..841bf08 100644 --- a/services/discord-gateway/src/modules/ai-moderation/moderationBuilders.ts +++ b/services/discord-gateway/src/modules/ai-moderation/moderationBuilders.ts @@ -7,6 +7,7 @@ import { messageStore } from "../message-capture/messageStore.js"; import type { MessageRecord } from "../message-capture/types.js"; +import { sanitizeDiscordTokens } from "./discordTokens.js"; /** Simple XML-escaping for content text. */ export function escapeXml(s: string): string { @@ -29,7 +30,7 @@ export function getAnalysisContent(message: MessageRecord): string { /\[(?:Attachment|Sticker):[^\]]*\]|\[Embed\]/g, "", ); - return stripped.trim(); + return sanitizeDiscordTokens(stripped).trim(); } /** diff --git a/services/discord-gateway/src/modules/ai-moderation/simpleFallback.ts b/services/discord-gateway/src/modules/ai-moderation/simpleFallback.ts deleted file mode 100644 index af2ace2..0000000 --- a/services/discord-gateway/src/modules/ai-moderation/simpleFallback.ts +++ /dev/null @@ -1,182 +0,0 @@ -/** - * simpleFallback.ts - * - * Simple two-step text fallback for cheap/small models. - * Step 1: Single-word classification (clean/warn/flagged). - * Step 2: Real analysis text (only if not clean). - * Extracted from moderationOrchestrator.ts. - */ -import { createChildLogger } from "@/shared/logger/index"; -import type { - AnalysisResult, - MessageRecord, -} from "../message-capture/types.js"; -import { llmChat } from "./llmClient.js"; -import { getAnalysisContent } from "./moderationBuilders.js"; -import { sanitizeAiContent } from "./moderationPrompt.js"; -import { getUserProfile } from "./userProfileStore.js"; - -const log = createChildLogger("simpleFallback"); - -// --------------------------------------------------------------------------- -// Simple text-only fallback -// --------------------------------------------------------------------------- - -/** - * Simple two-step text fallback for cheap/small models. - * Step 1: Single-word classification (clean/warn/flagged). - * Step 2: Real analysis text (only if not clean). - */ -export async function runSimpleTextFallback( - message: MessageRecord, -): Promise { - const content = getAnalysisContent(message); - const MAX_CONTENT_CHARS = 500; - const truncatedContent = - content.length > MAX_CONTENT_CHARS - ? `${content.slice(0, MAX_CONTENT_CHARS)}...` - : content; - - let userProfileCtx = ""; - try { - const profile = await getUserProfile(message.user_id); - if (profile?.profile_summary) { - userProfileCtx = `\n\nProfil pengirim pesan:\n${sanitizeAiContent(profile.profile_summary, 3000, false)}\n`; - } - } catch { - /* non-fatal */ - } - - // Step 1: Single-word classification - const classifyPrompt = `Pesan berikut perlu diklasifikasikan sebagai: clean, warn, atau flagged. - -Aturan: -- clean: pesan biasa, percakapan normal, tidak ada pelanggaran -- warn: spam ringan, promosi tidak jelas, atau pelanggaran ringan -- flagged: harassment, SARA, NSFW, judi, ancaman, atau pelanggaran serius - -PENTING (False Positive Prevention): -- Slang Indonesia ("anjay", "wkwk", "njir", "gws", dll) dan makian umum ("asu", "anjing", "bangsat") yang TIDAK ditujukan ke orang lain = clean. -- Konten coding/programming (kode, log error, SQL, command line, error message, stack trace, nama library) = clean. JANGAN flag hanya karena ada kata "error" atau "crash" dalam konteks teknis. -- Nama proyek, tools, framework (IMPHNEN, Bete, Cursor, Claude, React, Discord) = clean. -- Percakapan multilingual (campuran Indonesia-Inggris) = clean. -${userProfileCtx} -Pesan: "${truncatedContent}" - -Jawab HANYA dengan satu kata: clean, warn, atau flagged`; - - let status: "clean" | "warn" | "flagged"; - try { - const completion = await llmChat({ - messages: [{ role: "user", content: classifyPrompt }], - max_tokens: 10, - temperature: 0.1, - }); - const raw = - completion?.choices[0]?.message?.content?.trim().toLowerCase() ?? ""; - if (raw.includes("flagged")) status = "flagged"; - else if (raw.includes("warn")) status = "warn"; - else status = "clean"; - log.info({ messageId: message.id, status, raw }, "Simple fallback step 1"); - } catch (error) { - log.warn( - { - messageId: message.id, - error: error instanceof Error ? error.message : String(error), - }, - "Simple fallback step 1 failed — defaulting to clean", - ); - status = "clean"; - } - - // Step 2: Reason + category (only if not clean) - let analysis: string; - let category = ""; - - if (status === "clean") { - analysis = `${message.username ?? "user"}: ${content.length > 200 ? `${content.slice(0, 200)}...` : content}. Percakapan normal, tidak ada pelanggaran.`; - } else { - category = status === "flagged" ? "harassment" : "spam"; - const categoryOptions = - status === "flagged" ? "harassment, gambling, atau sara" : "spam"; - const reasonPrompt = `Pesan berikut telah diklasifikasikan sebagai "${status}". -${userProfileCtx} -Pesan: "${truncatedContent}" - -Jelaskan dalam 1-2 kalimat Bahasa Indonesia: APA yang melanggar dan KENAPA. Jangan gunakan kata "mungkin" atau "sepertinya". Jangan tulis ulang pesan. Langsung ke alasan. - -Setelah alasan, sebutkan Kategori: ${categoryOptions} - -Contoh untuk "flagged": -Mengandung kata kasar terarah ke individu tertentu sebagai hinaan. -Kategori: harassment - -Contoh untuk "flagged": -Promosi situs judi online dengan link dan ajakan. -Kategori: gambling - -Contoh untuk "warn": -Promosi channel Discord tanpa konteks, berpotensi spam. -Kategori: spam - -Contoh untuk "warn": -Bahasa kasar ringan yang tidak terarah. -Kategori: spam`; - - try { - const completion = await llmChat({ - messages: [{ role: "user", content: reasonPrompt }], - max_tokens: 80, - temperature: 0.3, - }); - analysis = completion?.choices[0]?.message?.content?.trim() ?? ""; - if (!analysis || analysis.length < 5) { - analysis = `Pesan diklasifikasikan sebagai ${status} oleh sistem moderasi otomatis.`; - } - const categoryMatch = analysis.match(/[Kk]ategori:\s*(\w+)/i); - if (categoryMatch) { - const parsedCat = categoryMatch[1].toLowerCase(); - if (["harassment", "spam", "gambling", "sara"].includes(parsedCat)) - category = parsedCat; - analysis = analysis.replace(/[Kk]ategori:\s*\w+\s*/i, "").trim(); - } - log.info( - { - messageId: message.id, - status, - category, - analysis: analysis.slice(0, 100), - }, - "Simple fallback step 2", - ); - } catch (error) { - analysis = `Pesan diklasifikasikan sebagai ${status} oleh sistem moderasi otomatis berdasarkan analisis konten.`; - log.warn( - { - messageId: message.id, - error: error instanceof Error ? error.message : String(error), - }, - "Simple fallback step 2 failed", - ); - } - } - - return { - messageId: message.id, - status, - flags: status === "clean" ? [] : [category], - score: status === "flagged" ? 0.7 : status === "warn" ? 0.4 : 0, - analysis, - categories: status === "clean" ? [] : [category], - severity: - status === "flagged" ? "medium" : status === "warn" ? "low" : "none", - confidence: 0.6, - recommendedAction: - status === "flagged" ? "review" : status === "warn" ? "warn" : "none", - policyVersion: "default-simple-2026-06", - evidence: - status !== "clean" - ? [content.length > 120 ? `${content.slice(0, 120)}...` : content] - : [], - }; -}