fix(gateway): surface failed URL fetches to moderation LLM
Links that fail to fetch (Facebook share 400, login walls, anti-bot) previously vanished from the prompt entirely — the LLM only saw the raw URL in the message body, so its analysis degraded to a template like "Pengguna membagikan tautan Facebook". Now the text batch tracks failed-fetch URLs and injects <web_content fetch_error="true"> telling the LLM the page content is unverifiable and to NOT invent it, plus a prompt rule: describe what the user actually expressed from the text + conversation context, or admit the subject is unclear — never make the link itself the subject.
This commit is contained in:
File diff suppressed because one or more lines are too long
@@ -50,6 +50,8 @@ interface UrlFetchResult {
|
|||||||
text: Map<string, string>;
|
text: Map<string, string>;
|
||||||
image: Map<string, { data: Buffer; mimeType: string }>;
|
image: Map<string, { data: Buffer; mimeType: string }>;
|
||||||
title: Map<string, string>;
|
title: Map<string, string>;
|
||||||
|
/** URL → reason the fetch failed (HTTP status / throw / unsupported). */
|
||||||
|
error: Map<string, string>;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Provider-reported token usage from a raw LLM payload (may be absent). */
|
/** Provider-reported token usage from a raw LLM payload (may be absent). */
|
||||||
@@ -169,6 +171,7 @@ export async function runTextOnlyBatch(
|
|||||||
text: new Map(),
|
text: new Map(),
|
||||||
image: new Map(),
|
image: new Map(),
|
||||||
title: new Map(),
|
title: new Map(),
|
||||||
|
error: new Map(),
|
||||||
} satisfies UrlFetchResult;
|
} satisfies UrlFetchResult;
|
||||||
}
|
}
|
||||||
const results = await Promise.allSettled(
|
const results = await Promise.allSettled(
|
||||||
@@ -177,9 +180,13 @@ export async function runTextOnlyBatch(
|
|||||||
const textMap = new Map<string, string>();
|
const textMap = new Map<string, string>();
|
||||||
const imageMap = new Map<string, { data: Buffer; mimeType: string }>();
|
const imageMap = new Map<string, { data: Buffer; mimeType: string }>();
|
||||||
const titleMap = new Map<string, string>();
|
const titleMap = new Map<string, string>();
|
||||||
|
const errorMap = new Map<string, string>();
|
||||||
for (let i = 0; i < urlArr.length; i++) {
|
for (let i = 0; i < urlArr.length; i++) {
|
||||||
const r = results[i];
|
const r = results[i];
|
||||||
if (r.status !== "fulfilled") continue;
|
if (r.status !== "fulfilled") {
|
||||||
|
errorMap.set(urlArr[i], "fetch threw");
|
||||||
|
continue;
|
||||||
|
}
|
||||||
const v = r.value;
|
const v = r.value;
|
||||||
if (v.type === "text" && v.textContent) {
|
if (v.type === "text" && v.textContent) {
|
||||||
textMap.set(urlArr[i], v.textContent);
|
textMap.set(urlArr[i], v.textContent);
|
||||||
@@ -188,9 +195,11 @@ export async function runTextOnlyBatch(
|
|||||||
// Direct image link (or og:image followed from an HTML page) —
|
// Direct image link (or og:image followed from an HTML page) —
|
||||||
// kept for vision analysis below.
|
// kept for vision analysis below.
|
||||||
imageMap.set(urlArr[i], { data: v.data, mimeType: v.mimeType });
|
imageMap.set(urlArr[i], { data: v.data, mimeType: v.mimeType });
|
||||||
|
} else {
|
||||||
|
errorMap.set(urlArr[i], v.error ?? "unsupported content");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return { text: textMap, image: imageMap, title: titleMap };
|
return { text: textMap, image: imageMap, title: titleMap, error: errorMap };
|
||||||
})();
|
})();
|
||||||
|
|
||||||
const webSearchPromise = (async () => {
|
const webSearchPromise = (async () => {
|
||||||
@@ -224,6 +233,7 @@ export async function runTextOnlyBatch(
|
|||||||
glossaryPromise,
|
glossaryPromise,
|
||||||
]);
|
]);
|
||||||
const urlFetchMap = urlFetchMaps.text;
|
const urlFetchMap = urlFetchMaps.text;
|
||||||
|
const urlFetchErrors = urlFetchMaps.error;
|
||||||
|
|
||||||
// Deduplicate identical short messages
|
// Deduplicate identical short messages
|
||||||
const shortContentGroups = new Map<string, MessageRecord[]>();
|
const shortContentGroups = new Map<string, MessageRecord[]>();
|
||||||
@@ -368,10 +378,16 @@ export async function runTextOnlyBatch(
|
|||||||
const urlContexts = msgUrls
|
const urlContexts = msgUrls
|
||||||
.map((url) => {
|
.map((url) => {
|
||||||
const ft = urlFetchMap.get(url);
|
const ft = urlFetchMap.get(url);
|
||||||
if (!ft) return null;
|
if (ft) {
|
||||||
const title = urlTitles.get(url);
|
const title = urlTitles.get(url);
|
||||||
const titleAttr = title ? ` title="${escapeXml(title)}"` : "";
|
const titleAttr = title ? ` title="${escapeXml(title)}"` : "";
|
||||||
return `<web_content url="${escapeXml(url)}"${titleAttr}>${escapeXml(ft)}</web_content>`;
|
return `<web_content url="${escapeXml(url)}"${titleAttr}>${escapeXml(ft)}</web_content>`;
|
||||||
|
}
|
||||||
|
const fetchError = urlFetchErrors.get(url);
|
||||||
|
if (fetchError) {
|
||||||
|
return `<web_content url="${escapeXml(url)}" fetch_error="true">Konten tidak dapat diambil otomatis (${escapeXml(fetchError)}). Analisis hanya dari teks pesan; JANGAN mengarang isi halaman.</web_content>`;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
})
|
})
|
||||||
.filter(Boolean)
|
.filter(Boolean)
|
||||||
.join("\n");
|
.join("\n");
|
||||||
|
|||||||
Reference in New Issue
Block a user