fix(gateway): flagged high/critical severity always eligible for auto-delete

The LLM's recommended_action is conservative — for a flagged message at
high/critical severity it frequently emits 'review' (screenshot/context
ambiguity) even when the violation is severe (harassment, SARA,
threats). autoDeleteEligibility trusted that value, so serious violations
slipped through undeleted (e.g. harassment flagged high but
review → not eligible).

Fix: flagged + high/critical severity bypasses the recommended-action
check entirely (always eligible); the action check now only gates
warn/flagged-medium. deriveRecommendedAction also returns delete for
flagged high/critical BEFORE consulting the stored LLM action. +5
regression tests.
This commit is contained in:
asepharyana
2026-09-24 18:57:41 +07:00
parent 54d02098c8
commit c888f23901
2 changed files with 104 additions and 10 deletions
@@ -35,8 +35,17 @@ export function deriveSeverity(msg: MessageRecord): string {
/** Derive recommended action from legacy messages that lack structured AI fields. */
export function deriveRecommendedAction(msg: MessageRecord): string {
if (msg.ai_recommended_action) return msg.ai_recommended_action;
const severity = deriveSeverity(msg);
// Flagged at high/critical severity is ALWAYS delete — the stored
// recommended_action from the LLM is conservative (often "review") and
// must not override severity for severe violations.
if (
msg.ai_status === "flagged" &&
(severity === "critical" || severity === "high")
) {
return "delete";
}
if (msg.ai_recommended_action) return msg.ai_recommended_action;
if (
msg.ai_status === "flagged" &&
(severity === "critical" || severity === "high" || severity === "medium")
@@ -177,7 +186,23 @@ export function isEligibleForAutoDelete(
return false;
}
// Recommended action check
// Recommended action check.
// CRITICAL: the LLM's `recommended_action` is CONSERVATIVE — for a flagged
// message at high/critical severity it frequently emits "review" (it sees a
// screenshot/context ambiguity and hedges) even when the violation itself is
// severe. Trusting that value lets serious violations (harassment, SARA,
// threats) slip through undeleted. So: flagged + high/critical severity is
// ALWAYS eligible regardless of the LLM's recommended action. The action
// check only gates warn/flagged-medium (where a review is legitimate).
if (
status === "flagged" &&
(severity === "high" || severity === "critical")
) {
logger.debug(
{ messageId: message.id, status, severity },
"Message eligible for auto-delete: flagged with high/critical severity",
);
} else {
const recommendedAction =
analysisResult?.recommendedAction ?? deriveRecommendedAction(message);
if (
@@ -191,6 +216,7 @@ export function isEligibleForAutoDelete(
);
return false;
}
}
// Categories check
const allowedCategories = parseStringList(
@@ -0,0 +1,68 @@
import { describe, expect, it } from "vitest";
import {
deriveRecommendedAction,
isEligibleForAutoDelete,
} from "../src/modules/ai-moderation/autoDeleteEligibility.js";
import type { MessageRecord } from "../src/shared/moderation-types.js";
const baseMessage: MessageRecord = {
id: "test-msg-1",
channel_id: "chan-1",
thread_id: null,
guild_id: "guild-1",
user_id: "user-1",
username: "tester",
content: "kau ngehina aku hitam kah ?",
created_at: Date.now(),
ai_status: "flagged",
ai_severity: "high",
ai_recommended_action: "review",
ai_moderation_flags: '["harassment"]',
ai_confidence: 0.95,
ai_analysis: "konfrontatif",
ai_categories: null,
ai_moderation_score: null,
ai_analyzed_at: Date.now(),
deleted_at: null,
} as unknown as MessageRecord;
describe("autoDeleteEligibility — flagged high severity with conservative LLM action", () => {
it("flagged + high severity is eligible even when LLM said review", () => {
const eligible = isEligibleForAutoDelete(baseMessage);
expect(eligible).toBe(true);
});
it("flagged + high severity derives delete regardless of stored review action", () => {
expect(deriveRecommendedAction(baseMessage)).toBe("delete");
});
it("flagged + medium severity with review action is NOT eligible", () => {
const medium = {
...baseMessage,
ai_severity: "medium",
} as unknown as MessageRecord;
const eligible = isEligibleForAutoDelete(medium);
expect(eligible).toBe(false);
});
it("warn status with review action is NOT eligible", () => {
const warn = {
...baseMessage,
ai_status: "warn",
ai_severity: "low",
ai_recommended_action: "warn",
} as unknown as MessageRecord;
const eligible = isEligibleForAutoDelete(warn);
expect(eligible).toBe(true); // warn action is allowed
});
it("clean status is never eligible", () => {
const clean = {
...baseMessage,
ai_status: "clean",
ai_severity: "none",
ai_recommended_action: "none",
} as unknown as MessageRecord;
expect(isEligibleForAutoDelete(clean)).toBe(false);
});
});