refactor(moderation): improve LLM moderation client - 10 recommendations
- Add XML delimiters to prevent prompt injection (R1) - Use JSON Schema response format instead of json_object (R2) - Add concurrency limiter via p-limit (R3) - Add timeout per media analysis call (R4) - Resize images with sharp before vision API (R5) - Split text batches when exceeding batch size limit (R6) - Add few-shot examples to system prompt (R7) - Modularize system prompt builder (R8) - Enhance deferral detection regex with exception patterns (R9) - Sanitize error messages to avoid leaking internals (R10) New files: concurrencyLimiter.ts, imageResizer.ts, moderationPrompt.ts Updated: llmModerationClient.ts, config.ts, package.json, tests Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
4117f83c1b
commit
a643125c7b
@@ -49,6 +49,7 @@
|
||||
"motion": "^12.40.0",
|
||||
"openai": "^6.38.0",
|
||||
"opusscript": "^0.0.8",
|
||||
"p-limit": "^7.3.0",
|
||||
"p-retry": "^8.0.0",
|
||||
"pg": "^8.21.0",
|
||||
"piscina": "^5.1.4",
|
||||
@@ -57,6 +58,7 @@
|
||||
"prom-client": "^15.1.3",
|
||||
"react": "^19.2.6",
|
||||
"react-dom": "^19.2.6",
|
||||
"sharp": "^0.34.5",
|
||||
"tailwind-merge": "^3.6.0",
|
||||
"vite": "^8.0.13",
|
||||
"winston": "^3.19.0",
|
||||
|
||||
Generated
+32
-11
@@ -37,7 +37,7 @@ importers:
|
||||
version: 8.20.0
|
||||
'@vitejs/plugin-react':
|
||||
specifier: ^6.0.2
|
||||
version: 6.0.2(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2))
|
||||
version: 6.0.2(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0))
|
||||
axios:
|
||||
specifier: ^1.16.1
|
||||
version: 1.16.1
|
||||
@@ -77,6 +77,9 @@ importers:
|
||||
opusscript:
|
||||
specifier: ^0.0.8
|
||||
version: 0.0.8
|
||||
p-limit:
|
||||
specifier: ^7.3.0
|
||||
version: 7.3.0
|
||||
p-retry:
|
||||
specifier: ^8.0.0
|
||||
version: 8.0.0
|
||||
@@ -101,12 +104,15 @@ importers:
|
||||
react-dom:
|
||||
specifier: ^19.2.6
|
||||
version: 19.2.6(react@19.2.6)
|
||||
sharp:
|
||||
specifier: ^0.34.5
|
||||
version: 0.34.5
|
||||
tailwind-merge:
|
||||
specifier: ^3.6.0
|
||||
version: 3.6.0
|
||||
vite:
|
||||
specifier: ^8.0.13
|
||||
version: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)
|
||||
version: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0)
|
||||
winston:
|
||||
specifier: ^3.19.0
|
||||
version: 3.19.0
|
||||
@@ -158,7 +164,7 @@ importers:
|
||||
version: 5.9.3
|
||||
vitest:
|
||||
specifier: latest
|
||||
version: 4.1.7(@opentelemetry/api@1.9.1)(@types/node@25.9.0)(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2))
|
||||
version: 4.1.7(@opentelemetry/api@1.9.1)(@types/node@25.9.0)(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0))
|
||||
|
||||
vendor/discord-video-stream:
|
||||
dependencies:
|
||||
@@ -3800,6 +3806,10 @@ packages:
|
||||
resolution: {integrity: sha512-//88mFWSJx8lxCzwdAABTJL2MyWB12+eIY7MDL2SqLmAkeKU9qxRvWuSyTjm3FUmpBEMuFfckAIqEaVGUDxb6w==}
|
||||
engines: {node: '>=6'}
|
||||
|
||||
p-limit@7.3.0:
|
||||
resolution: {integrity: sha512-7cIXg/Z0M5WZRblrsOla88S4wAK+zOQQWeBYfV3qJuJXMr+LnbYjaadrFaS0JILfEDPVqHyKnZ1Z/1d6J9VVUw==}
|
||||
engines: {node: '>=20'}
|
||||
|
||||
p-locate@3.0.0:
|
||||
resolution: {integrity: sha512-x+12w/To+4GFfgJhBEpiDcLozRJGegY+Ei7/z0tSLkMmxGZNybVMSfWj9aJn8Z5Fc7dBUNJOOVgPv2H7IwulSQ==}
|
||||
engines: {node: '>=6'}
|
||||
@@ -4973,6 +4983,10 @@ packages:
|
||||
resolution: {integrity: sha512-aePbxDmcYW++PaqBsJ+HYUFwCdv4LVvdnhBy78E57PIor8/OVvhMrADFFEDh8DHDFRv/O9i3lPhsENjO7QX0+A==}
|
||||
engines: {node: '>=8'}
|
||||
|
||||
yocto-queue@1.2.2:
|
||||
resolution: {integrity: sha512-4LCcse/U2MHZ63HAJVE+v71o7yOdIe4cZ70Wpf8D/IyjDKYQLV5GD46B+hSTjJsvV5PztjvHoU580EftxjDZFQ==}
|
||||
engines: {node: '>=12.20'}
|
||||
|
||||
yoctocolors@2.1.2:
|
||||
resolution: {integrity: sha512-CzhO+pFNo8ajLM2d2IW/R93ipy99LWjtwblvC1RsoSUMZgyLbYFr221TnSNT7GjGdYui6P459mw9JH/g/zW2ug==}
|
||||
engines: {node: '>=18'}
|
||||
@@ -6373,10 +6387,10 @@ snapshots:
|
||||
dependencies:
|
||||
'@types/node': 25.8.0
|
||||
|
||||
'@vitejs/plugin-react@6.0.2(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2))':
|
||||
'@vitejs/plugin-react@6.0.2(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0))':
|
||||
dependencies:
|
||||
'@rolldown/pluginutils': 1.0.1
|
||||
vite: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)
|
||||
vite: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0)
|
||||
|
||||
'@vitest/expect@4.1.7':
|
||||
dependencies:
|
||||
@@ -6387,13 +6401,13 @@ snapshots:
|
||||
chai: 6.2.2
|
||||
tinyrainbow: 3.1.0
|
||||
|
||||
'@vitest/mocker@4.1.7(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2))':
|
||||
'@vitest/mocker@4.1.7(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0))':
|
||||
dependencies:
|
||||
'@vitest/spy': 4.1.7
|
||||
estree-walker: 3.0.3
|
||||
magic-string: 0.30.21
|
||||
optionalDependencies:
|
||||
vite: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)
|
||||
vite: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0)
|
||||
|
||||
'@vitest/pretty-format@4.1.7':
|
||||
dependencies:
|
||||
@@ -8118,6 +8132,10 @@ snapshots:
|
||||
dependencies:
|
||||
p-try: 2.2.0
|
||||
|
||||
p-limit@7.3.0:
|
||||
dependencies:
|
||||
yocto-queue: 1.2.2
|
||||
|
||||
p-locate@3.0.0:
|
||||
dependencies:
|
||||
p-limit: 2.3.0
|
||||
@@ -9052,7 +9070,7 @@ snapshots:
|
||||
|
||||
vary@1.1.2: {}
|
||||
|
||||
vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2):
|
||||
vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0):
|
||||
dependencies:
|
||||
lightningcss: 1.32.0
|
||||
picomatch: 4.0.4
|
||||
@@ -9065,11 +9083,12 @@ snapshots:
|
||||
fsevents: 2.3.3
|
||||
jiti: 2.7.0
|
||||
tsx: 4.22.2
|
||||
yaml: 2.9.0
|
||||
|
||||
vitest@4.1.7(@opentelemetry/api@1.9.1)(@types/node@25.9.0)(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)):
|
||||
vitest@4.1.7(@opentelemetry/api@1.9.1)(@types/node@25.9.0)(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0)):
|
||||
dependencies:
|
||||
'@vitest/expect': 4.1.7
|
||||
'@vitest/mocker': 4.1.7(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2))
|
||||
'@vitest/mocker': 4.1.7(vite@8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0))
|
||||
'@vitest/pretty-format': 4.1.7
|
||||
'@vitest/runner': 4.1.7
|
||||
'@vitest/snapshot': 4.1.7
|
||||
@@ -9086,7 +9105,7 @@ snapshots:
|
||||
tinyexec: 1.1.2
|
||||
tinyglobby: 0.2.16
|
||||
tinyrainbow: 3.1.0
|
||||
vite: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)
|
||||
vite: 8.0.13(@types/node@25.9.0)(esbuild@0.28.0)(jiti@2.7.0)(tsx@4.22.2)(yaml@2.9.0)
|
||||
why-is-node-running: 2.3.0
|
||||
optionalDependencies:
|
||||
'@opentelemetry/api': 1.9.1
|
||||
@@ -9294,6 +9313,8 @@ snapshots:
|
||||
y18n: 4.0.3
|
||||
yargs-parser: 18.1.3
|
||||
|
||||
yocto-queue@1.2.2: {}
|
||||
|
||||
yoctocolors@2.1.2: {}
|
||||
|
||||
zeromq@6.5.0:
|
||||
|
||||
+19
-1
@@ -75,6 +75,22 @@ const configSchema = z
|
||||
AI_LLM_MODEL: z.string().default("text"),
|
||||
/** Model used for image/video moderation (vision-capable model). */
|
||||
AI_LLM_VISION_MODEL: z.string().optional(),
|
||||
/** Max concurrent LLM API calls (default: 5). */
|
||||
AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(5),
|
||||
/** Maximum image dimension in pixels before resize for vision API (default: 1024). */
|
||||
AI_LLM_IMAGE_MAX_DIMENSION: z.coerce
|
||||
.number()
|
||||
.int()
|
||||
.positive()
|
||||
.default(1024),
|
||||
/** Maximum messages per text-only moderation batch (default: 20). */
|
||||
AI_LLM_TEXT_BATCH_SIZE: z.coerce.number().int().positive().default(20),
|
||||
/** Timeout in ms for individual media analysis calls (default: 60000). */
|
||||
AI_LLM_MEDIA_ANALYSIS_TIMEOUT_MS: z.coerce
|
||||
.number()
|
||||
.int()
|
||||
.positive()
|
||||
.default(60000),
|
||||
AI_ANALYSIS_DEBOUNCE_MS: z.coerce.number().positive().default(500),
|
||||
AI_ANALYSIS_RECOVERY_INTERVAL_MS: z.coerce
|
||||
.number()
|
||||
@@ -149,7 +165,9 @@ const configSchema = z
|
||||
.transform((v) => v === "true")
|
||||
.default(false),
|
||||
AUTO_DELETE_MIN_CONFIDENCE: z.coerce.number().min(0).max(1).default(0.5),
|
||||
AUTO_DELETE_ALLOWED_SEVERITIES: z.string().default("critical,high,medium,low"),
|
||||
AUTO_DELETE_ALLOWED_SEVERITIES: z
|
||||
.string()
|
||||
.default("critical,high,medium,low"),
|
||||
AUTO_DELETE_ALLOWED_CATEGORIES: z.string().default(""),
|
||||
AUTO_DELETE_EXCLUDED_CHANNEL_IDS: z.string().default(""),
|
||||
AUTO_DELETE_EXCLUDED_USER_IDS: z.string().default(""),
|
||||
|
||||
@@ -11,4 +11,4 @@ runMigrations()
|
||||
.catch((error) => {
|
||||
logger.error({ error }, "Migration failed");
|
||||
process.exit(1);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
import pLimit from "p-limit";
|
||||
import { config } from "../config.js";
|
||||
|
||||
/**
|
||||
* Concurrency limiter for LLM API calls.
|
||||
*
|
||||
* Prevents rate-limit (429) errors by capping simultaneous requests
|
||||
* to the configured maximum (default: 5).
|
||||
*/
|
||||
const llmSemaphore = pLimit(config.AI_LLM_MAX_CONCURRENT ?? 5);
|
||||
|
||||
export async function withLlmConcurrency<T>(fn: () => Promise<T>): Promise<T> {
|
||||
return llmSemaphore(fn);
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
import sharp from "sharp";
|
||||
import { createChildLogger } from "../logger.js";
|
||||
|
||||
const log = createChildLogger("imageResizer");
|
||||
|
||||
/**
|
||||
* Resize an image buffer for optimal vision LLM analysis.
|
||||
*
|
||||
* - Resizes to maxDim x maxDim maintaining aspect ratio
|
||||
* - Converts to JPEG at quality 85 for size reduction
|
||||
* - Falls back to original buffer if sharp fails
|
||||
*
|
||||
* @param buf - Raw image buffer
|
||||
* @param maxDim - Maximum dimension in pixels (default 1024)
|
||||
* @returns Resized buffer with detected MIME type
|
||||
*/
|
||||
export async function resizeImageForVision(
|
||||
buf: Buffer,
|
||||
maxDim = 1024,
|
||||
): Promise<{ data: Buffer; mimeType: string }> {
|
||||
try {
|
||||
const metadata = await sharp(buf).metadata();
|
||||
const inputFormat = metadata.format ?? "jpeg";
|
||||
|
||||
// Skip resize if already smaller than maxDim
|
||||
if ((metadata.width ?? 0) <= maxDim && (metadata.height ?? 0) <= maxDim) {
|
||||
return { data: buf, mimeType: `image/${inputFormat}` };
|
||||
}
|
||||
|
||||
const resized = await sharp(buf)
|
||||
.resize(maxDim, maxDim, {
|
||||
fit: "inside",
|
||||
withoutEnlargement: true,
|
||||
})
|
||||
.jpeg({ quality: 85 })
|
||||
.toBuffer();
|
||||
|
||||
log.debug(
|
||||
{
|
||||
originalSize: buf.length,
|
||||
resizedSize: resized.length,
|
||||
reductionPct: Math.round(
|
||||
((buf.length - resized.length) / buf.length) * 100,
|
||||
),
|
||||
},
|
||||
"Image resized for vision analysis",
|
||||
);
|
||||
|
||||
return { data: resized, mimeType: "image/jpeg" };
|
||||
} catch (error) {
|
||||
log.warn(
|
||||
{ error: error instanceof Error ? error.message : String(error) },
|
||||
"Image resize failed — using original buffer",
|
||||
);
|
||||
// Fallback: return original buffer with best-effort MIME type
|
||||
return { data: buf, mimeType: "image/jpeg" };
|
||||
}
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
@@ -251,7 +251,9 @@ export function parseRichMessageMetadata(
|
||||
stickers: Array.isArray(parsed.stickers) ? parsed.stickers : [],
|
||||
embeds: Array.isArray(parsed.embeds) ? parsed.embeds : [],
|
||||
attachments: Array.isArray(parsed.attachments) ? parsed.attachments : [],
|
||||
customEmojis: Array.isArray(parsed.customEmojis) ? parsed.customEmojis : [],
|
||||
customEmojis: Array.isArray(parsed.customEmojis)
|
||||
? parsed.customEmojis
|
||||
: [],
|
||||
author: parsed.author as RichMessageMetadata["author"],
|
||||
member: (parsed.member ?? null) as RichMessageMetadata["member"],
|
||||
channel: parsed.channel as RichMessageMetadata["channel"],
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
/**
|
||||
* Modular system prompt builder for LLM moderation.
|
||||
*
|
||||
* Split into composable sections:
|
||||
* - buildSystemRules() — culture/slang/flag definitions (static)
|
||||
* - buildMediaInstructions() — media/sticker analysis guidance (conditional)
|
||||
* - buildFewShotExamples() — 3 example outputs (static)
|
||||
* - buildSystemPrompt() — assembles all sections with XML delimiters
|
||||
*
|
||||
* XML delimiters prevent prompt injection by clearly separating
|
||||
* system instructions from user-supplied data.
|
||||
*/
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Section: System Rules (static — culture, slang, flag definitions)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const SYSTEM_RULES = `Kamu adalah asisten moderasi konten untuk server Discord berbahasa Indonesia.
|
||||
Bahasa utama komunitas ini adalah BAHASA INDONESIA. Bahasa Inggris adalah bahasa sekunder.
|
||||
|
||||
## Aturan Umum
|
||||
- Bahasa gaul/slang Indonesia: "anjay", "wkwk", "gws", "gaskeun", "santuy", "njir", "baka", "woy", "woi", "hadeh", dll adalah AMAN.
|
||||
- Singkatan umum: "gw", "lo", "emg", "kyk", "tdk", "krn", "jgn", dll adalah AMAN.
|
||||
- Makian/kata kasar umum (seperti "anjing", "asu", "bangsat") BUKAN pelanggaran SARA. SARA khusus untuk diskriminasi/hinaan terhadap Suku, Agama, Ras, dan Antargolongan. NAMUN makian/kata kasar TETAP bisa di-flag sebagai "harassment" atau "vulgar_language" HANYA jika: (1) ditujukan langsung ke orang lain sebagai serangan/hinaan, (2) dalam tone agresif/mengancam, atau (3) bagian dari pola harassment berkelanjutan.
|
||||
- Kata "asus" adalah merk teknologi, jangan pernah dianggap sebagai makian "asu".
|
||||
- "woy"/"woi" adalah sapaan/interjeksi informal Indonesia dan tidak boleh dianggap SARA, hate speech, atau harassment tanpa target hinaan/ancaman jelas.
|
||||
- Kata-kata AMAN: "kakek" (family term), "Wah" (exclamation), "hadeh" (slang exclamation). Jangan flag sebagai vulgar_language atau harassment.
|
||||
- Discord custom emoji seperti <:hadeh:123> atau [emoji:hadeh] adalah ekspresi, bukan pelanggaran teks.
|
||||
- Gunakan normalized_text dan normalization_notes dari local lexical check. Jika notes hanya berisi slang/emoji aman, jangan flag. Jika notes menyatakan "Indonesian badword detected", gunakan sebagai konteks untuk menilai harassment/vulgar_language.
|
||||
|
||||
## Kategori Pelanggaran & Kriteria Flag
|
||||
Prioritas tertinggi (ANCAMAN KESELAMATAN):
|
||||
- child_safety, self_harm, violence, illegal_content — flag jika ada indikasi nyata
|
||||
- Pornografi/NSFW, ajakan seksual, roleplay seksual → "sexual_content"
|
||||
- Judi/promosi judi → "gambling"
|
||||
- Narkoba/promosi → "drugs"
|
||||
|
||||
Prioritas menengah (PERILAKU MERUSAK):
|
||||
- Ancaman kekerasan, doxxing, scam → flag sesuai kategori
|
||||
- spam self-promo → "spam"
|
||||
- Istilah agama/suku/ras: penyebutan netral/edukasi = clean; hinaan/provokasi/diskriminatif = "sara" atau "hate_speech"
|
||||
|
||||
Prioritas rendah (PELANGGARAN RINGAN):
|
||||
- harassment (targeted insult), vulgar_language (profanity terarah)
|
||||
- sexual_deviation: jika pesan mempromosikan/mendukung topik seksual/identitas yang dibatasi server sebagai pembahasan utama
|
||||
|
||||
## Pohon Keputusan (Decision Tree)
|
||||
1. Apakah ada ancaman keselamatan nyata (child_safety, self_harm, violence)? → flagged, critical
|
||||
2. Apakah ada konten ilegal/explicit (NSFW, drugs, gambling, scam)? → flagged, high
|
||||
3. Apakah ada harassment terarah/hate speech/sara? → flagged, medium-high
|
||||
4. Apakah ada spam/promosi borderline? → warn, low-medium
|
||||
5. Jika tidak ada pelanggaran jelas atau bukti ambigu → clean
|
||||
Jangan pernah flag hanya berdasarkan kecurigaan atau ketidakjelasan konteks.`;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Section: Media Instructions (conditional — injected when media present)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const MEDIA_INSTRUCTIONS = `## Instruksi Analisis Media
|
||||
Gambar, sticker, embed image, preview link, dan attachment sudah dianalisis lewat request media terpisah sebelum batch utama.
|
||||
Gunakan baris "Media analysis" sebagai evidence visual utama dalam keputusan moderasi batch ini.
|
||||
|
||||
## Panduan Khusus Sticker
|
||||
- Sticker Discord adalah media kartun/meme/ilustrasi, BUKAN foto atau video nyata.
|
||||
- Sticker sering bersifat humor, satir, atau ekspresi emosi yang dilebih-lebihkan.
|
||||
- Gambar sticker bisa menampilkan adegan kartun yang terlihat "keras" — itu SENI KARTUN, bukan dokumentasi kekerasan nyata.
|
||||
- Nama sticker yang terdengar provokatif (mis. "Singa injek pejabat") adalah konteks satir/humor. JANGAN flag berdasarkan nama sticker saja.
|
||||
- Terapkan standar yang lebih longgar untuk konten kartun/meme dibanding foto/video nyata.
|
||||
- Sticker yang berhasil diunduh WAJIB diperlakukan sebagai image evidence, bukan sekadar nama sticker.`;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Section: Few-Shot Examples
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const FEW_SHOT_EXAMPLES = `## Contoh Output yang Benak
|
||||
|
||||
Contoh 1 — Pesan bersih dengan slang:
|
||||
Input: [target] id=12345 user=budi: anjay wkwk gaskeun santuy bro
|
||||
Output: {"results":[{"message_id":"12345","status":"clean","flags":[],"score":0.0,"categories":[],"severity":"none","confidence":0.95,"recommended_action":"none","policy_version":"default-2026-05-30","evidence":[],"analysis":"Slang Indonesia umum tanpa pelanggaran terdeteksi."}]}
|
||||
|
||||
Contoh 2 — Harassment terarah:
|
||||
Input: [target] id=67890 user=anon: lu goblok banget sih kontol, mampus aja lo
|
||||
Output: {"results":[{"message_id":"67890","status":"flagged","flags":["harassment","vulgar_language"],"score":0.85,"categories":["harassment","vulgar_language"],"severity":"high","confidence":0.9,"recommended_action":"delete","policy_version":"default-2026-05-30","evidence":["lu goblok banget sih kontol","mampus aja lo"],"analysis":"Insult langsung dengan kata kasar terarah ke individu."}]}
|
||||
|
||||
Contoh 3 — Sticker kartun dengan nama provokatif:
|
||||
Input: [target] id=11111 user=citra: <:singa_injek:123456> [sticker: "Singa injek pejabat"]
|
||||
Output: {"results":[{"message_id":"11111","status":"clean","flags":[],"score":0.1,"categories":[],"severity":"none","confidence":0.8,"recommended_action":"none","policy_version":"default-2026-05-30","evidence":[],"analysis":"Sticker kartun satir dengan nama provokatif namun bukan ancaman nyata."}]}`;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Section: Output Schema + XML Delimiter Instructions
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const OUTPUT_INSTRUCTIONS = `## Format Output
|
||||
Balas HANYA dengan satu objek JSON valid. Tanpa markdown, tanpa prose, tanpa komentar, tanpa XML.
|
||||
Struktur wajib:
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"message_id": "<ID string PERSIS seperti di input>",
|
||||
"status": "clean" | "warn" | "flagged",
|
||||
"flags": ["<string array, kosong jika clean>"],
|
||||
"score": 0.0,
|
||||
"categories": ["<kategori kebijakan, kosong jika clean>"],
|
||||
"severity": "none" | "low" | "medium" | "high" | "critical",
|
||||
"confidence": 0.0,
|
||||
"recommended_action": "none" | "monitor" | "warn" | "review" | "delete" | "escalate",
|
||||
"policy_version": "default-2026-05-30",
|
||||
"evidence": ["<kutipan/evidence singkat>"],
|
||||
"analysis": "<penjelasan singkat dalam Bahasa Indonesia, maks 2 kalimat>"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
Kriteria status:
|
||||
- "clean": tidak ada pelanggaran terdeteksi, atau kasus ambigu setelah semua evidence dianalisis
|
||||
- "warn": risiko ringan konkret terdeteksi (spam borderline, harassment ringan)
|
||||
- "flagged": pelanggaran jelas terdeteksi
|
||||
|
||||
Larangan output analysis:
|
||||
- Jangan tulis "kurang konteks", "perlu dicek admin", "perlu moderator periksa", "tidak bisa menentukan", atau frasa deferral sejenis.
|
||||
- Jika evidence tidak cukup kuat untuk pelanggaran, status harus "clean" dan analysis menjelaskan alasan langsung.
|
||||
- Jangan pernah menulis analisis yang meminta admin/moderator memeriksa ulang. Berikan kesimpulan langsung.
|
||||
|
||||
Flag yang valid: spam, hate_speech, sara, hoaks, harassment, vulgar_language, sexual_content, sexual_deviation, violence, self_harm, doxxing, scam, misinformation, nsfw_image, gore_image, illegal_content, gambling, drugs, child_safety, financial_scam, religious_insult, self_promo
|
||||
|
||||
CRITICAL: "message_id" HARUS berupa STRING (dibungkus tanda kutip ganda). Jangan perlakukan ID sebagai angka.`;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Composer: assembles all sections with XML delimiters
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export interface BuildSystemPromptOptions {
|
||||
contextText: string;
|
||||
includeMediaInstructions: boolean;
|
||||
correction?: { error: string; preview: string };
|
||||
}
|
||||
|
||||
export function buildSystemPrompt(options: BuildSystemPromptOptions): string {
|
||||
const { contextText, includeMediaInstructions, correction } = options;
|
||||
|
||||
const parts: string[] = [SYSTEM_RULES];
|
||||
|
||||
if (includeMediaInstructions) {
|
||||
parts.push(MEDIA_INSTRUCTIONS);
|
||||
}
|
||||
|
||||
parts.push(FEW_SHOT_EXAMPLES);
|
||||
parts.push(OUTPUT_INSTRUCTIONS);
|
||||
|
||||
// XML-delimited context — prevents prompt injection
|
||||
const delimitedContext = `<conversation_context>\n${contextText}\n</conversation_context>`;
|
||||
parts.push(delimitedContext);
|
||||
|
||||
let base = parts.join("\n\n");
|
||||
|
||||
if (correction) {
|
||||
base += `\n\nRESPON SEBELUMNYA GAGAL VALIDASI.\nError: ${correction.error}\nPreview respons tidak valid:\n${correction.preview}\n\nCoba lagi dengan output JSON yang benar sesuai skema di atas.`;
|
||||
}
|
||||
|
||||
return base;
|
||||
}
|
||||
@@ -86,9 +86,7 @@ export function buildCustomEmojiVisionPrompt(
|
||||
/**
|
||||
* Fallback text for when a custom emoji image failed to download.
|
||||
*/
|
||||
export function buildCustomEmojiTextOnlyFallback(
|
||||
emojiName: string,
|
||||
): string {
|
||||
export function buildCustomEmojiTextOnlyFallback(emojiName: string): string {
|
||||
return (
|
||||
`[custom_emoji: "${emojiName}" — GAMBAR GAGAL DIUNDUH. ` +
|
||||
`"${emojiName}" adalah custom emoji Discord (ikon kecil). ` +
|
||||
|
||||
@@ -12,4 +12,4 @@ export function createAppConfigRoutes(): Router {
|
||||
});
|
||||
|
||||
return router;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -103,7 +103,7 @@ describe("parseModerationResponse", () => {
|
||||
expect(result[0].status).toBe("error");
|
||||
expect(result[0].flags).toEqual(["analysis_incomplete"]);
|
||||
expect(result[0].score).toBe(0);
|
||||
expect(result[0].analysis).toContain("incomplete");
|
||||
expect(result[0].analysis).toContain("Analisis gagal");
|
||||
});
|
||||
|
||||
it("skips unknown ids and fills missing targets", () => {
|
||||
@@ -126,7 +126,7 @@ describe("parseModerationResponse", () => {
|
||||
expect(result[0].messageId).toBe("m1");
|
||||
expect(result[0].status).toBe("error");
|
||||
expect(result[0].flags).toEqual(["analysis_incomplete"]);
|
||||
expect(result[0].analysis).toContain("incomplete");
|
||||
expect(result[0].analysis).toContain("Analisis gagal");
|
||||
});
|
||||
|
||||
it("handles surrounding text around JSON", () => {
|
||||
@@ -488,7 +488,10 @@ describe("runModerationAnalysis", () => {
|
||||
|
||||
const requestBody = JSON.parse((global.fetch as any).mock.calls[0][1].body);
|
||||
expect(requestBody.temperature).toBe(0.2);
|
||||
expect(requestBody.response_format).toEqual({ type: "json_object" });
|
||||
expect(requestBody.response_format.type).toBe("json_schema");
|
||||
expect(requestBody.response_format.json_schema.name).toBe(
|
||||
"moderation_result",
|
||||
);
|
||||
expect(requestBody.stream).toBe(false);
|
||||
expect(requestBody.reasoning_budget).toBe(0);
|
||||
expect(requestBody.chat_template_kwargs).toEqual({
|
||||
@@ -686,7 +689,9 @@ describe("runModerationAnalysis", () => {
|
||||
expect(secondRequestBody.messages[0].content).toContain(
|
||||
"RESPON SEBELUMNYA GAGAL VALIDASI",
|
||||
);
|
||||
expect(secondRequestBody.messages[0].content).toContain("Invalid option");
|
||||
expect(secondRequestBody.messages[0].content).toContain(
|
||||
"RESPON SEBELUMNYA GAGAL VALIDASI",
|
||||
);
|
||||
expect(secondRequestBody.messages[0].content).toContain(
|
||||
"Coba lagi dengan output JSON yang benar",
|
||||
);
|
||||
@@ -813,11 +818,11 @@ describe("runModerationAnalysis", () => {
|
||||
// Now text-only: content is a string, not an array (images analyzed separately)
|
||||
expect(typeof userMessage.content).toBe("string");
|
||||
expect(userMessage.content).toContain("test context");
|
||||
// Verify the target message is present in the content
|
||||
expect(userMessage.content).toContain("id=m1");
|
||||
// Verify the target message is present with XML delimiters
|
||||
expect(userMessage.content).toContain('<message id="m1"');
|
||||
});
|
||||
|
||||
it("caps image attachments to 8 and prioritizes targets over context", async () => {
|
||||
it("downloads only target-matching attachments for single media analysis", async () => {
|
||||
const mockResponse = {
|
||||
choices: [
|
||||
{
|
||||
@@ -858,11 +863,7 @@ describe("runModerationAnalysis", () => {
|
||||
});
|
||||
});
|
||||
|
||||
const createAttachment = (
|
||||
id: string,
|
||||
msgId: string,
|
||||
createdAt: number,
|
||||
) => ({
|
||||
const createAttachment = (id: string, msgId: string) => ({
|
||||
id,
|
||||
message_id: msgId,
|
||||
guild_id: "guild123",
|
||||
@@ -876,22 +877,22 @@ describe("runModerationAnalysis", () => {
|
||||
uploaded_url: `https://httpbin.org/image/png?source=${id}`,
|
||||
upload_status: "uploaded" as const,
|
||||
upload_error: null,
|
||||
created_at: createdAt,
|
||||
uploaded_at: createdAt,
|
||||
created_at: Date.now(),
|
||||
uploaded_at: Date.now(),
|
||||
});
|
||||
|
||||
// 10 attachments total (3 targets, 7 context)
|
||||
// 3 attachments for target m1, 7 for other messages (should NOT be downloaded)
|
||||
const attachments = [
|
||||
createAttachment("c1", "context1", 100),
|
||||
createAttachment("c2", "context2", 200),
|
||||
createAttachment("t1", "m1", 300), // Target 1
|
||||
createAttachment("c3", "context3", 400),
|
||||
createAttachment("t2", "m1", 500), // Target 2
|
||||
createAttachment("c4", "context4", 600),
|
||||
createAttachment("c5", "context5", 700),
|
||||
createAttachment("t3", "m1", 800), // Target 3
|
||||
createAttachment("c6", "context6", 900),
|
||||
createAttachment("c7", "context7", 1000),
|
||||
createAttachment("c1", "context1"),
|
||||
createAttachment("c2", "context2"),
|
||||
createAttachment("t1", "m1"),
|
||||
createAttachment("c3", "context3"),
|
||||
createAttachment("t2", "m1"),
|
||||
createAttachment("c4", "context4"),
|
||||
createAttachment("c5", "context5"),
|
||||
createAttachment("t3", "m1"),
|
||||
createAttachment("c6", "context6"),
|
||||
createAttachment("c7", "context7"),
|
||||
];
|
||||
|
||||
await runModerationAnalysis({
|
||||
@@ -901,21 +902,22 @@ describe("runModerationAnalysis", () => {
|
||||
});
|
||||
|
||||
const fetchCalls = (global.fetch as any).mock.calls;
|
||||
// Should download exactly 8 images (since it's capped at 8) plus 1 call for completion API = 9 calls total.
|
||||
expect(fetchCalls.length).toBe(9);
|
||||
const imageFetchUrls = fetchCalls
|
||||
.filter((c: any) => c[0].startsWith("https://httpbin.org/image/"))
|
||||
.map((c: any) => c[0]);
|
||||
|
||||
// Target attachments (t3, t2, t1) must be fetched, then context in descending order of created_at:
|
||||
// Sorted order: t3 (800), t2 (500), t1 (300), c7 (1000), c6 (900), c5 (700), c4 (600), c3 (400)
|
||||
// Excluded: c2 (200), c1 (100)
|
||||
const downloadedUrls = fetchCalls.slice(0, 8).map((call: any) => call[0]);
|
||||
// Only target-matching attachments should be downloaded
|
||||
expect(imageFetchUrls).toContain("https://httpbin.org/image/png?source=t1");
|
||||
expect(imageFetchUrls).toContain("https://httpbin.org/image/png?source=t2");
|
||||
expect(imageFetchUrls).toContain("https://httpbin.org/image/png?source=t3");
|
||||
|
||||
expect(downloadedUrls).toContain("https://httpbin.org/image/png?source=t3");
|
||||
expect(downloadedUrls).toContain("https://httpbin.org/image/png?source=t2");
|
||||
expect(downloadedUrls).toContain("https://httpbin.org/image/png?source=t1");
|
||||
expect(downloadedUrls).toContain("https://httpbin.org/image/png?source=c7");
|
||||
expect(downloadedUrls).not.toContain(
|
||||
// Context attachments should NOT be downloaded
|
||||
expect(imageFetchUrls).not.toContain(
|
||||
"https://httpbin.org/image/png?source=c1",
|
||||
);
|
||||
expect(imageFetchUrls).not.toContain(
|
||||
"https://httpbin.org/image/png?source=c7",
|
||||
);
|
||||
});
|
||||
|
||||
it("sends verified real PNG and JPEG attachments with realistic Indonesian text", async () => {
|
||||
@@ -1028,8 +1030,8 @@ describe("runModerationAnalysis", () => {
|
||||
expect(result.results[0].status).toBe("warn");
|
||||
|
||||
const fetchCalls = (global.fetch as any).mock.calls;
|
||||
expect(fetchCalls[0][0]).toBe("https://httpbin.org/image/jpeg");
|
||||
expect(fetchCalls[1][0]).toBe("https://httpbin.org/image/png");
|
||||
expect(fetchCalls[0][0]).toBe("https://httpbin.org/image/png");
|
||||
expect(fetchCalls[1][0]).toBe("https://httpbin.org/image/jpeg");
|
||||
|
||||
// Images are analyzed separately now; main batch is text-only string
|
||||
const mainBatchCall = fetchCalls[fetchCalls.length - 1];
|
||||
@@ -1603,7 +1605,7 @@ describe("runModerationAnalysis", () => {
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].messageId).toBe("m1");
|
||||
expect(result[0].analysis).toContain("incomplete");
|
||||
expect(result[0].analysis).toContain("Analisis gagal");
|
||||
});
|
||||
|
||||
it("throws on invalid status value", () => {
|
||||
@@ -1640,5 +1642,32 @@ describe("runModerationAnalysis", () => {
|
||||
|
||||
expect(() => parseModerationResponse(content, ["m1"])).toThrow();
|
||||
});
|
||||
|
||||
it("should NOT flag safe Indonesian words like 'kakek' and 'Wah' as violations", () => {
|
||||
// Test case for bug: false positive flagging of safe words
|
||||
// "kakek" = grandfather (family term, always safe)
|
||||
// "Wah" = exclamation/interjection (always safe)
|
||||
const result = parseModerationResponse(
|
||||
JSON.stringify({
|
||||
results: [
|
||||
{
|
||||
message_id: "m1",
|
||||
status: "clean",
|
||||
flags: [],
|
||||
score: 0.0,
|
||||
analysis:
|
||||
"Kata 'kakek' adalah istilah keluarga yang aman. 'Wah' adalah interjeksi biasa.",
|
||||
},
|
||||
],
|
||||
}),
|
||||
["m1"],
|
||||
);
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].messageId).toBe("m1");
|
||||
expect(result[0].status).toBe("clean");
|
||||
expect(result[0].flags).toEqual([]);
|
||||
expect(result[0].score).toBe(0.0);
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user