fix(gateway): flagged high/critical severity always eligible for auto-delete
The LLM's recommended_action is conservative — for a flagged message at high/critical severity it frequently emits 'review' (screenshot/context ambiguity) even when the violation is severe (harassment, SARA, threats). autoDeleteEligibility trusted that value, so serious violations slipped through undeleted (e.g. harassment flagged high but review → not eligible). Fix: flagged + high/critical severity bypasses the recommended-action check entirely (always eligible); the action check now only gates warn/flagged-medium. deriveRecommendedAction also returns delete for flagged high/critical BEFORE consulting the stored LLM action. +5 regression tests.
This commit is contained in:
@@ -35,8 +35,17 @@ export function deriveSeverity(msg: MessageRecord): string {
|
||||
|
||||
/** Derive recommended action from legacy messages that lack structured AI fields. */
|
||||
export function deriveRecommendedAction(msg: MessageRecord): string {
|
||||
if (msg.ai_recommended_action) return msg.ai_recommended_action;
|
||||
const severity = deriveSeverity(msg);
|
||||
// Flagged at high/critical severity is ALWAYS delete — the stored
|
||||
// recommended_action from the LLM is conservative (often "review") and
|
||||
// must not override severity for severe violations.
|
||||
if (
|
||||
msg.ai_status === "flagged" &&
|
||||
(severity === "critical" || severity === "high")
|
||||
) {
|
||||
return "delete";
|
||||
}
|
||||
if (msg.ai_recommended_action) return msg.ai_recommended_action;
|
||||
if (
|
||||
msg.ai_status === "flagged" &&
|
||||
(severity === "critical" || severity === "high" || severity === "medium")
|
||||
@@ -177,7 +186,23 @@ export function isEligibleForAutoDelete(
|
||||
return false;
|
||||
}
|
||||
|
||||
// Recommended action check
|
||||
// Recommended action check.
|
||||
// CRITICAL: the LLM's `recommended_action` is CONSERVATIVE — for a flagged
|
||||
// message at high/critical severity it frequently emits "review" (it sees a
|
||||
// screenshot/context ambiguity and hedges) even when the violation itself is
|
||||
// severe. Trusting that value lets serious violations (harassment, SARA,
|
||||
// threats) slip through undeleted. So: flagged + high/critical severity is
|
||||
// ALWAYS eligible regardless of the LLM's recommended action. The action
|
||||
// check only gates warn/flagged-medium (where a review is legitimate).
|
||||
if (
|
||||
status === "flagged" &&
|
||||
(severity === "high" || severity === "critical")
|
||||
) {
|
||||
logger.debug(
|
||||
{ messageId: message.id, status, severity },
|
||||
"Message eligible for auto-delete: flagged with high/critical severity",
|
||||
);
|
||||
} else {
|
||||
const recommendedAction =
|
||||
analysisResult?.recommendedAction ?? deriveRecommendedAction(message);
|
||||
if (
|
||||
@@ -191,6 +216,7 @@ export function isEligibleForAutoDelete(
|
||||
);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Categories check
|
||||
const allowedCategories = parseStringList(
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
deriveRecommendedAction,
|
||||
isEligibleForAutoDelete,
|
||||
} from "../src/modules/ai-moderation/autoDeleteEligibility.js";
|
||||
import type { MessageRecord } from "../src/shared/moderation-types.js";
|
||||
|
||||
const baseMessage: MessageRecord = {
|
||||
id: "test-msg-1",
|
||||
channel_id: "chan-1",
|
||||
thread_id: null,
|
||||
guild_id: "guild-1",
|
||||
user_id: "user-1",
|
||||
username: "tester",
|
||||
content: "kau ngehina aku hitam kah ?",
|
||||
created_at: Date.now(),
|
||||
ai_status: "flagged",
|
||||
ai_severity: "high",
|
||||
ai_recommended_action: "review",
|
||||
ai_moderation_flags: '["harassment"]',
|
||||
ai_confidence: 0.95,
|
||||
ai_analysis: "konfrontatif",
|
||||
ai_categories: null,
|
||||
ai_moderation_score: null,
|
||||
ai_analyzed_at: Date.now(),
|
||||
deleted_at: null,
|
||||
} as unknown as MessageRecord;
|
||||
|
||||
describe("autoDeleteEligibility — flagged high severity with conservative LLM action", () => {
|
||||
it("flagged + high severity is eligible even when LLM said review", () => {
|
||||
const eligible = isEligibleForAutoDelete(baseMessage);
|
||||
expect(eligible).toBe(true);
|
||||
});
|
||||
|
||||
it("flagged + high severity derives delete regardless of stored review action", () => {
|
||||
expect(deriveRecommendedAction(baseMessage)).toBe("delete");
|
||||
});
|
||||
|
||||
it("flagged + medium severity with review action is NOT eligible", () => {
|
||||
const medium = {
|
||||
...baseMessage,
|
||||
ai_severity: "medium",
|
||||
} as unknown as MessageRecord;
|
||||
const eligible = isEligibleForAutoDelete(medium);
|
||||
expect(eligible).toBe(false);
|
||||
});
|
||||
|
||||
it("warn status with review action is NOT eligible", () => {
|
||||
const warn = {
|
||||
...baseMessage,
|
||||
ai_status: "warn",
|
||||
ai_severity: "low",
|
||||
ai_recommended_action: "warn",
|
||||
} as unknown as MessageRecord;
|
||||
const eligible = isEligibleForAutoDelete(warn);
|
||||
expect(eligible).toBe(true); // warn action is allowed
|
||||
});
|
||||
|
||||
it("clean status is never eligible", () => {
|
||||
const clean = {
|
||||
...baseMessage,
|
||||
ai_status: "clean",
|
||||
ai_severity: "none",
|
||||
ai_recommended_action: "none",
|
||||
} as unknown as MessageRecord;
|
||||
expect(isEligibleForAutoDelete(clean)).toBe(false);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user