refactor(ai): replace SearXNG with Wikipedia adapter for analysis enrichment

- Add wikipediaClient.ts: native fetch to Wikipedia REST/Action APIs
  (search + summary), no extra npm dependency.
- Extract shared Redis cache into cacheStore.ts (decoupled from search).
- Term glossary now uses wikipediaSummary for direct article lookup.
- Remove searxngSearch.ts entirely; drop SEARXNG_BASE_URL config,
  add WIKIPEDIA_LANG / WIKIPEDIA_TIMEOUT_MS.
- Rename backend searxngCalls metric to webSearchCalls.
This commit is contained in:
asepharyana
2026-08-17 20:07:19 +07:00
parent 2825250804
commit 479f4719ba
10 changed files with 434 additions and 589 deletions
@@ -67,9 +67,9 @@ export const moderationErrors = new Counter({
labelNames: ["type"] as const,
});
export const searxngCalls = new Counter({
name: "moderation_searxng_calls_total",
help: "SearXNG search calls",
export const webSearchCalls = new Counter({
name: "moderation_websearch_calls_total",
help: "Wikipedia web-search calls",
labelNames: ["status"] as const,
});
+36 -200
View File
@@ -29,9 +29,6 @@ importers:
drizzle-orm:
specifier: ^0.45.2
version: 0.45.2(@types/pg@8.20.0)(pg@8.22.0)
imghash:
specifier: ^1.1.4
version: 1.1.4
ioredis:
specifier: ^5.11.0
version: 5.11.1(supports-color@7.2.0)
@@ -80,7 +77,7 @@ importers:
devDependencies:
'@biomejs/biome':
specifier: latest
version: 2.5.7
version: 2.5.8
'@types/node':
specifier: ^25.9.0
version: 25.9.5
@@ -105,86 +102,63 @@ importers:
packages:
'@biomejs/biome@2.5.7':
resolution: {integrity: sha512-zr8K/DcY5tYsQOQwqMJ0AWElo6QgmgNI7idXgXLhevVszlt8RGVpesEJPqx3ThazLaOwjJ5Y8fz3BtH5fGZNsw==}
'@biomejs/biome@2.5.8':
resolution: {integrity: sha512-aeAeeJB9fSDc7Gq+2GqpQxA0qBj6gj1k2R6L1cYqGePKP/baIq1WX8y6B+D+nRsO5ViQL22K/8IwbqERW0q1nw==}
engines: {node: '>=14.21.3'}
hasBin: true
'@biomejs/cli-darwin-arm64@2.5.7':
resolution: {integrity: sha512-vxo/Ls3/PYdQWyLhYYcgMOCzQypAjcY+iihS8M0wW03l16TCLW4zqZzGo75gm1VdCMj38hTVZ31KBWrZ4G9dJw==}
'@biomejs/cli-darwin-arm64@2.5.8':
resolution: {integrity: sha512-mk1QON9PHllvvLN5gU3f4rMxeh4syK5p9OvKyWH6/W8ueh04uaC8TUXXByhGufWf/y5mQc03ZLM45zU+cmqMjA==}
engines: {node: '>=14.21.3'}
cpu: [arm64]
os: [darwin]
'@biomejs/cli-darwin-x64@2.5.7':
resolution: {integrity: sha512-Cd3Ga61amT/Yl/0x8elP5hhGYaFy4bw6WuysTgf7oo8TA5tJ5A1k+DkVoJ2BHbTVil51gTX9VPzArnrlLJ3Kyg==}
'@biomejs/cli-darwin-x64@2.5.8':
resolution: {integrity: sha512-bsGwFMBNyHPyiLSsQcZJxdoRrg1V4JL+d7wEsvUBczlP9U9lwM+7mzQHxI4o1mhBsTmdOBbAb6fHU3Z3snN45w==}
engines: {node: '>=14.21.3'}
cpu: [x64]
os: [darwin]
'@biomejs/cli-linux-arm64-musl@2.5.7':
resolution: {integrity: sha512-xPI5yB6XlpDbNkS+bm1t42olw5c4l3UrlOmLg7KtLJvjvkNF/1V4tnUgfkylGIeb3u/T+BzMGYqgQhzjAoJzuQ==}
'@biomejs/cli-linux-arm64-musl@2.5.8':
resolution: {integrity: sha512-VcJNbstduTHx83NGAdhp78/JOcP45BZHXL7yNsfI1uGzdUgegAz2s+mSoT7wK6PBNzLoqG0zDOXaz/RQYVtSiw==}
engines: {node: '>=14.21.3'}
cpu: [arm64]
os: [linux]
libc: [musl]
'@biomejs/cli-linux-arm64@2.5.7':
resolution: {integrity: sha512-rR2QE0yF2GYSuYuKIa7pKvODGJqnOH+2eDREAM8wV+mWKSkMQKdAp4zXEZfTaxY8PMoNONnpgSWcBCyLDPDOKg==}
'@biomejs/cli-linux-arm64@2.5.8':
resolution: {integrity: sha512-XmFiA0WPYFC+uiUDC8WRFzAIH9bo7vwQLav38Uoq4ETC+T/+uBi0TsYGJECkugY3r8USl3jc+Ae2/irAF6F2lQ==}
engines: {node: '>=14.21.3'}
cpu: [arm64]
os: [linux]
libc: [glibc]
'@biomejs/cli-linux-x64-musl@2.5.7':
resolution: {integrity: sha512-rE5VZi+qtmPgQH+l7jVxYoZ18b/TiHEhulhMpjmCZH1PltSbjRcxNWywC3HZ9tYottG7ORkeTtoscBilKSBm0g==}
'@biomejs/cli-linux-x64-musl@2.5.8':
resolution: {integrity: sha512-kKmiyokeISRGq2FLwvr+TzsgBusfxaZ0FZNLcOYOpCK/78tRrEjeEBLvq3xLZMpqbANgJdRPI7vZX8ZL37u9/w==}
engines: {node: '>=14.21.3'}
cpu: [x64]
os: [linux]
libc: [musl]
'@biomejs/cli-linux-x64@2.5.7':
resolution: {integrity: sha512-FQgqJhscrqJUFptGaRSUJWlXAExwWcDwLuK49dvKfkQ1bB5SEEyFssnsxQY83Xm6jR0EbbX3+8+D5bfvYqUG2Q==}
'@biomejs/cli-linux-x64@2.5.8':
resolution: {integrity: sha512-S5wcm9OBDvLHodD4PUaN488hCpco9QD/9ZxuYJiw4euWtr/oQvLR72z2ixItH8Wd5BCm6FZaeb+YNvOoM1xHtQ==}
engines: {node: '>=14.21.3'}
cpu: [x64]
os: [linux]
libc: [glibc]
'@biomejs/cli-win32-arm64@2.5.7':
resolution: {integrity: sha512-Oq4x0CCwP4jirrcTywXs5kOGZ4v5vuEP+gWrbtjApOA2CL9F3F9GlIdQIci8AKSCa/zURanMRpX/4wQ7Am6hHg==}
'@biomejs/cli-win32-arm64@2.5.8':
resolution: {integrity: sha512-nILH0mzm3Hi3iEdd7o7GpB8kBR/mSQwfQG/tyBqyNrY2GFtcgwfV9nV8xLmbtUpMNY/Oi0Ml1XgfR4flOdq+AA==}
engines: {node: '>=14.21.3'}
cpu: [arm64]
os: [win32]
'@biomejs/cli-win32-x64@2.5.7':
resolution: {integrity: sha512-V+0wu/nrj2S+MhP4EQ0uHNolP0IALEsz45pg0WoKkHfDeh0+ItHwP/p7bX5RPoMOl9NkpHYWdYPhIcy2mACHvQ==}
'@biomejs/cli-win32-x64@2.5.8':
resolution: {integrity: sha512-I2czzXTY61f3nFJxXoMDq80t7MivxDEnCjE+8sDKoFfcKMaoQdkqhIFQ3KyY0XLzeSpUBYeNAXgD+iOV/BU0VA==}
engines: {node: '>=14.21.3'}
cpu: [x64]
os: [win32]
'@canvas/image-data@1.1.0':
resolution: {integrity: sha512-QdObRRjRbcXGmM1tmJ+MrHcaz1MftF2+W7YI+MsphnsCrmtyfS0d5qJbk0MeSbUeyM/jCb0hmnkXPsy026L7dA==}
'@canvas/image@2.0.0':
resolution: {integrity: sha512-DQKEftZ5M4eM8Rzhv8FFZGdOsO8bVjzPC/9pBF+P//HjQM/Xf+PvQRTv97mVpsHt+Pejs1V7ooK6xf+dYgDNpw==}
engines: {node: '>=10'}
'@cwasm/jpeg-turbo@0.1.3':
resolution: {integrity: sha512-FkZxwwC6r4zhzlqM0nYGaMj/MDSrZPxLOdPdM6ySlgsMfOpNAZcLQkpNF4jP+DmsuUvRoeUD0YSMBvg3jYfK6w==}
'@cwasm/lodepng@0.1.9':
resolution: {integrity: sha512-vb2H7/jTxnqJi7hHiEgtFm3smIqVpeY417vN+8cwsq3iTNrHzwnMzFXbeCf2H9Dl13Vh64qgUUcC2mxf6TPODA==}
engines: {node: '>=8.0.0'}
'@cwasm/nsbmp@0.1.3':
resolution: {integrity: sha512-APiz9Rj2E049rBapTtwnCGqQeqJjmC85busDQ44UCaujU+LoggUm06NuS1WIBAZcDMRbJOgCFcRWc0tMT2kpfg==}
'@cwasm/nsgif@0.1.2':
resolution: {integrity: sha512-LOD5HlL0O5jpnIAl+dLSZcB3v0RBNBjtoaymdCEPe2kyKzaP20BF+jy/QUyOZogQsgMVjusZES3tgwwoiiJ2rA==}
'@cwasm/webp@0.1.5':
resolution: {integrity: sha512-ceIZQkyxK+s7mmItNcWqqHdOBiJAxYxTnrnPNgUNjldB1M9j+Bp/3eVIVwC8rUFyN/zoFwuT0331pyY3ackaNA==}
'@discordjs/builders@1.14.1':
resolution: {integrity: sha512-gSKkhXLqs96TCzk66VZuHHl8z2bQMJFGwrXC0f33ngK+FLNau4hU1PYny3DNJfNdSH+gVMzE85/d5FQ2BpcNwQ==}
engines: {node: '>=16.11.0'}
@@ -1288,9 +1262,6 @@ packages:
base64-js@1.5.1:
resolution: {integrity: sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA==}
blockhash-core@0.1.0:
resolution: {integrity: sha512-Cv7BgBo0jjVPaeuel4cvxf9LqIGsYNIPz9DAGvvrF9LRlEq9Q3HXu+S8bklPCae0sCxAXic4HGMoImf3FeO3Nw==}
brace-expansion@1.1.18:
resolution: {integrity: sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==}
@@ -1377,18 +1348,6 @@ packages:
resolution: {integrity: sha512-z2S+W9X73hAUUki+N+9Za2lBlun89zigOyGrsax+KUQ6wKW4ZoWpEYBkGhQjwAjjDCkWxhY0VKEhk8wzY7F5cA==}
engines: {node: '>=0.10.0'}
decode-bmp@0.2.1:
resolution: {integrity: sha512-NiOaGe+GN0KJqi2STf24hfMkFitDUaIoUU3eKvP/wAbLe8o6FuW5n/x7MHPR0HKvBokp6MQY/j7w8lewEeVCIA==}
engines: {node: '>=8.6.0'}
decode-ico@0.4.1:
resolution: {integrity: sha512-69NZfbKIzux1vBOd31al3XnMnH+2mqDhEgLdpygErm4d60N+UwA5Sq5WFjmEDQzumgB9fElojGwWG0vybVfFmA==}
engines: {node: '>=8.6'}
decompress-response@6.0.0:
resolution: {integrity: sha512-aW35yZM6Bb/4oJlZncMH2LCoZtJXTRxES17vE3hoRiowU2kWHaJKFkSBDnDR+cm9J+9QhXmREyIfv0pji9ejCQ==}
engines: {node: '>=10'}
delayed-stream@1.0.0:
resolution: {integrity: sha512-ZySD7Nf91aLB0RxL4KGrKHBXl7Eds1DAmEdcoVawXnLD7SDhpNgtuII2aAkg7a7QS41jxPSZ17p4VdGnMHk3MQ==}
engines: {node: '>=0.4.0'}
@@ -1566,15 +1525,6 @@ packages:
resolution: {integrity: sha512-KfYbmpRm0VbLjEvVa9yGwCi9GI34xvi7A/HXYWQO65CSD2u3MczUJSuwXKFIxlGsgBQizV9q5J9NHj4VG0n+pA==}
engines: {node: '>=12.0.0'}
fast-base64-decode@1.0.0:
resolution: {integrity: sha512-qwaScUgUGBYeDNRnbc/KyllVU88Jk1pRHPStuF/lO7B0/RTRLj7U0lkdTAutlBblY08rwZDff6tNU9cjv6j//Q==}
fast-base64-encode@1.0.0:
resolution: {integrity: sha512-z2XCzVK4fde2cuTEHu2QGkLD6BPtJNKJPn0Z7oINvmhq/quUuIIVPYKUdN0gYeZqOyurjJjBH/bUzK5gafyHvw==}
fast-base64-length@1.0.0:
resolution: {integrity: sha512-MV+/ioblHx6SMjc/1l4EAnRJyAku6+6DxZ6RW0FoFCF1Aol/Ldb6FqwE3Kn3Ju1aam2m1KCIVoCljhgcG+Umzg==}
fast-deep-equal@3.1.3:
resolution: {integrity: sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q==}
@@ -1590,10 +1540,6 @@ packages:
fetch-cookie@3.2.0:
resolution: {integrity: sha512-n61pQIxP25C6DRhcJxn7BDzgHP/+S56Urowb5WFxtcRMpU6drqXD90xjyAsVQYsNSNNVbaCcYY1DuHsdkZLuiA==}
file-type@10.11.0:
resolution: {integrity: sha512-uzk64HRpUZyTGZtVuvrjP0FYxzQrBf4rojot6J65YMEbwBLB0CWm0CLojVpwpmFmxcE/lkvYICgfcGozbBq6rw==}
engines: {node: '>=6'}
find-process@2.1.1:
resolution: {integrity: sha512-SrQDx3QhlmHM90iqn9rdjCQcw/T+WlpOkHFsjoRgB+zTpDfltNA1VSNYeYELwhUTJy12UFxqjWhmhOrJc+o4sA==}
hasBin: true
@@ -1684,14 +1630,6 @@ packages:
ieee754@1.2.1:
resolution: {integrity: sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA==}
image-type@4.1.0:
resolution: {integrity: sha512-CFJMJ8QK8lJvRlTCEgarL4ro6hfDQKif2HjSvYCdQZESaIPV4v9imrf7BQHK+sQeTeNeMpWciR9hyC/g8ybXEg==}
engines: {node: '>=6'}
imghash@1.1.4:
resolution: {integrity: sha512-atRI6S7rwrdZlr8pBsuk0NfBfeuPzyna98ugz2Sodj1P8SOZJLVYuqWnUrUFg2YnFTxtdbdSq4Ucn1osLFQ+qA==}
engines: {node: '>=20'}
inflight@1.0.6:
resolution: {integrity: sha512-k92I/b08q4wvFscXCLvqfsHCrjrF7yiXsQuIVvVE7N82W3+aqpzuUdBbfhWcy/FZR3/4IgflMgKLOsvPDrGCJA==}
deprecated: This module is not supported, and leaks memory. Do not use it. Check out lru-cache if you want a good and tested way to coalesce async requests by a key value, which is much more comprehensive and powerful.
@@ -1711,9 +1649,6 @@ packages:
resolution: {integrity: sha512-PhBY86zaxNZUuWP6h13Vu5oFe0XY6/UlKzQnYFELzGVHygP3MxmvTfYSG7GN3aIab/iWudSMgjSnG9Dq+nHrgA==}
engines: {node: '>=16'}
jpeg-js@0.4.4:
resolution: {integrity: sha512-WZzeDOEtTOBK4Mdsar0IqEU5sMr3vSV2RqkAIzUEV2BHnUfKGyswWFPFwK5EeDo93K3FohSHbLAjj0s1Wzd+dg==}
libsodium-wrappers@0.8.4:
resolution: {integrity: sha512-mu8aAWucZjTB5O/BtGXtW4e1agy7uHxNYG7zPthmmD1jU43LCDmSWZLN4JhflbdPXj3yDO4lxM1O9hLDgIOXDw==}
@@ -1831,10 +1766,6 @@ packages:
resolution: {integrity: sha512-ZDY+bPm5zTTF+YpCrAU9nK0UgICYPT0QtT1NZWFv4s++TNkcgVaT0g6+4R2uI4MjQjzysHB1zxuWL50hzaeXiw==}
engines: {node: '>= 0.6'}
mimic-response@3.1.0:
resolution: {integrity: sha512-z0yWI+4FDrrweS8Zmt4Ej5HdJmky15+L2e6Wgn3+iK5fWzb6T3fhNFq2+MeTRb064c6Wr4N/wv0DzQTjNzHNGQ==}
engines: {node: '>=10'}
minimatch@3.1.5:
resolution: {integrity: sha512-VgjWUsnnT6n+NUk6eZq77zeFdpW2LWDzP6zFGrCbHXiYNul5Dzqk2HHQ5uFH2DNW5Xbp8+jVzaeNt94ssEEl4w==}
@@ -2139,12 +2070,6 @@ packages:
signal-exit@3.0.7:
resolution: {integrity: sha512-wnD2ZE+l+SPC/uoS0vXeE9L1+0wuaMqKlfz9AMUo38JsyLSBWSFcHR1Rri62LZc12vLr1gb3jl7iwQhgwpAbGQ==}
simple-concat@1.0.1:
resolution: {integrity: sha512-cSFtAPtRhljv69IK0hTVZQ+OfE9nePi/rtJmw5UjHeVyVroEqJXP1sFztKUy1qU+xvz3u/sfYJLa947b7nAN2Q==}
simple-get@4.0.1:
resolution: {integrity: sha512-brv7p5WgH0jmQJr1ZDDfKDOSeWWg+OVypG99A/5vYGPqJ6pxiaHLy8nxtFjBA7oMa01ebA9gfh1uMCFqOuXxvA==}
sonic-boom@4.2.1:
resolution: {integrity: sha512-w6AxtubXa2wTXAUsZMMWERrsIRAdrK0Sc+FUytWvYAhBJLyuI4llrMIC1DtlNSdI99EI86KZum2MMq3EAZlF9Q==}
@@ -2231,9 +2156,6 @@ packages:
resolution: {integrity: sha512-3kZ8wQQ/k5DrChD4X4FVvr2D7E5uoRgAqkPyLpSCGUvqOvqu+JEdr3mwMUaVWb+vMHZaKhF5fp2PBigKsui7hA==}
hasBin: true
to-data-view@1.1.0:
resolution: {integrity: sha512-1eAdufMg6mwgmlojAx3QeMnzB/BTVp7Tbndi3U7ftcT2zCZadjxkkmLmd97zmaxWi+sgGcgWrokmpEoy0Dn0vQ==}
tough-cookie@5.1.2:
resolution: {integrity: sha512-FVDYdxtnj0G6Qm/DhNPSb8Ju59ULcup3tuJxkFb5K8Bv2pUXILbf0xZWU8PX8Ov19OXljbUyveOFwRMwkXzO+A==}
engines: {node: '>=16'}
@@ -2426,77 +2348,41 @@ packages:
snapshots:
'@biomejs/biome@2.5.7':
'@biomejs/biome@2.5.8':
optionalDependencies:
'@biomejs/cli-darwin-arm64': 2.5.7
'@biomejs/cli-darwin-x64': 2.5.7
'@biomejs/cli-linux-arm64': 2.5.7
'@biomejs/cli-linux-arm64-musl': 2.5.7
'@biomejs/cli-linux-x64': 2.5.7
'@biomejs/cli-linux-x64-musl': 2.5.7
'@biomejs/cli-win32-arm64': 2.5.7
'@biomejs/cli-win32-x64': 2.5.7
'@biomejs/cli-darwin-arm64': 2.5.8
'@biomejs/cli-darwin-x64': 2.5.8
'@biomejs/cli-linux-arm64': 2.5.8
'@biomejs/cli-linux-arm64-musl': 2.5.8
'@biomejs/cli-linux-x64': 2.5.8
'@biomejs/cli-linux-x64-musl': 2.5.8
'@biomejs/cli-win32-arm64': 2.5.8
'@biomejs/cli-win32-x64': 2.5.8
'@biomejs/cli-darwin-arm64@2.5.7':
'@biomejs/cli-darwin-arm64@2.5.8':
optional: true
'@biomejs/cli-darwin-x64@2.5.7':
'@biomejs/cli-darwin-x64@2.5.8':
optional: true
'@biomejs/cli-linux-arm64-musl@2.5.7':
'@biomejs/cli-linux-arm64-musl@2.5.8':
optional: true
'@biomejs/cli-linux-arm64@2.5.7':
'@biomejs/cli-linux-arm64@2.5.8':
optional: true
'@biomejs/cli-linux-x64-musl@2.5.7':
'@biomejs/cli-linux-x64-musl@2.5.8':
optional: true
'@biomejs/cli-linux-x64@2.5.7':
'@biomejs/cli-linux-x64@2.5.8':
optional: true
'@biomejs/cli-win32-arm64@2.5.7':
'@biomejs/cli-win32-arm64@2.5.8':
optional: true
'@biomejs/cli-win32-x64@2.5.7':
'@biomejs/cli-win32-x64@2.5.8':
optional: true
'@canvas/image-data@1.1.0': {}
'@canvas/image@2.0.0':
dependencies:
'@canvas/image-data': 1.1.0
'@cwasm/jpeg-turbo': 0.1.3
'@cwasm/lodepng': 0.1.9
'@cwasm/nsbmp': 0.1.3
'@cwasm/nsgif': 0.1.2
'@cwasm/webp': 0.1.5
decode-ico: 0.4.1
fast-base64-decode: 1.0.0
fast-base64-encode: 1.0.0
fast-base64-length: 1.0.0
simple-get: 4.0.1
'@cwasm/jpeg-turbo@0.1.3':
dependencies:
'@canvas/image-data': 1.1.0
'@cwasm/lodepng@0.1.9':
dependencies:
'@canvas/image-data': 1.1.0
'@cwasm/nsbmp@0.1.3':
dependencies:
'@canvas/image-data': 1.1.0
'@cwasm/nsgif@0.1.2':
dependencies:
'@canvas/image-data': 1.1.0
'@cwasm/webp@0.1.5':
dependencies:
'@canvas/image-data': 1.1.0
'@discordjs/builders@1.14.1':
dependencies:
'@discordjs/formatters': 0.6.2
@@ -3280,8 +3166,6 @@ snapshots:
base64-js@1.5.1: {}
blockhash-core@0.1.0: {}
brace-expansion@1.1.18:
dependencies:
balanced-match: 1.0.2
@@ -3352,21 +3236,6 @@ snapshots:
decamelize@1.2.0: {}
decode-bmp@0.2.1:
dependencies:
'@canvas/image-data': 1.1.0
to-data-view: 1.1.0
decode-ico@0.4.1:
dependencies:
'@canvas/image-data': 1.1.0
decode-bmp: 0.2.1
to-data-view: 1.1.0
decompress-response@6.0.0:
dependencies:
mimic-response: 3.1.0
delayed-stream@1.0.0: {}
delegates@1.0.0: {}
@@ -3535,12 +3404,6 @@ snapshots:
expect-type@1.4.0: {}
fast-base64-decode@1.0.0: {}
fast-base64-encode@1.0.0: {}
fast-base64-length@1.0.0: {}
fast-deep-equal@3.1.3: {}
fdir@6.5.0(picomatch@4.0.5):
@@ -3552,8 +3415,6 @@ snapshots:
set-cookie-parser: 2.7.2
tough-cookie: 6.0.2
file-type@10.11.0: {}
find-process@2.1.1:
dependencies:
chalk: 4.1.2
@@ -3658,17 +3519,6 @@ snapshots:
ieee754@1.2.1: {}
image-type@4.1.0:
dependencies:
file-type: 10.11.0
imghash@1.1.4:
dependencies:
'@canvas/image': 2.0.0
blockhash-core: 0.1.0
image-type: 4.1.0
jpeg-js: 0.4.4
inflight@1.0.6:
dependencies:
once: 1.4.0
@@ -3692,8 +3542,6 @@ snapshots:
is-network-error@1.3.2: {}
jpeg-js@0.4.4: {}
libsodium-wrappers@0.8.4:
dependencies:
libsodium: 0.8.4
@@ -3780,8 +3628,6 @@ snapshots:
dependencies:
mime-db: 1.52.0
mimic-response@3.1.0: {}
minimatch@3.1.5:
dependencies:
brace-expansion: 1.1.18
@@ -4061,14 +3907,6 @@ snapshots:
signal-exit@3.0.7: {}
simple-concat@1.0.1: {}
simple-get@4.0.1:
dependencies:
decompress-response: 6.0.0
once: 1.4.0
simple-concat: 1.0.1
sonic-boom@4.2.1:
dependencies:
atomic-sleep: 1.0.0
@@ -4148,8 +3986,6 @@ snapshots:
dependencies:
tldts-core: 7.4.9
to-data-view@1.1.0: {}
tough-cookie@5.1.2:
dependencies:
tldts: 6.1.86
@@ -0,0 +1,78 @@
/**
* cacheStore.ts
*
* Shared Redis cache used by the AI-moderation modules (term glossary, etc.).
*
* Extracted when SearXNG was removed (replaced by the Wikipedia adapter in
* wikipediaClient.ts). The cache was never SearXNG-specific — it is a generic
* namespaced key/value store with graceful degradation when Redis is
* unavailable. Other modules import `makeCacheKey`, `cacheGet`, `cacheSet`,
* and `initCacheStore` instead of reaching into a search module.
*/
import Redis from "ioredis";
import { createChildLogger } from "@/shared/logger/index";
const log = createChildLogger("cache-store");
const CACHE_PREFIX = "gmw:";
const CACHE_TTL = 86400; // 24 hours (used as a sane default)
let redis: Redis | null = null;
/**
* Initialize the shared Redis connection for the moderation cache.
* Safe to call multiple times — only creates one connection.
* Degrades gracefully to `null` (no-cache) when Redis is unavailable.
*/
export function initCacheStore(redisUrl: string): void {
if (redis) return;
redis = new Redis(redisUrl, {
maxRetriesPerRequest: 3,
retryStrategy(times) {
const delay = Math.min(times * 200, 2000);
return delay;
},
lazyConnect: true,
enableReadyCheck: false,
});
redis.on("error", (err) => {
log.warn({ err: err.message }, "Cache Redis error");
});
redis.connect().catch(() => {
log.warn("Cache Redis unavailable — falling back to no-cache");
redis = null;
});
log.info("Cache Redis initialized");
}
/** Exposes the shared Redis connection; null when Redis is unavailable. */
export function getCacheRedis(): Redis | null {
return redis;
}
/** Builds a namespaced cache key (shared across modules). */
export function makeCacheKey(namespace: string, key: string): string {
return `${CACHE_PREFIX}${namespace}:${key.toLowerCase().trim()}`;
}
/** Reads a value from the cache; null on miss/unavailable. */
export async function cacheGet(key: string): Promise<string | null> {
if (!redis) return null;
try {
return await redis.get(key);
} catch {
return null;
}
}
/** Writes a value to the cache, fire-and-forget. */
export function cacheSet(key: string, value: string, ttlSeconds: number): void {
if (!redis) return;
redis.setex(key, ttlSeconds, value).catch(() => {
// Cache write failed silently
});
}
/** Default TTL (exposed for callers that want the standard window). */
export const DEFAULT_CACHE_TTL = CACHE_TTL;
@@ -12,12 +12,12 @@ import type {
AttachmentRecord,
MessageRecord,
} from "../message-capture/types.js";
import { initCacheStore } from "./cacheStore.js";
import { embedTexts, isEmbeddingEnabled } from "./embeddingClient.js";
import { hasMediaContent } from "./mediaAnalysisClient.js";
import { runMediaBatch } from "./mediaBatchProcessor.js";
import { isQdrantConfigured, searchQdrantBatch } from "./qdrantClient.js";
import { logCacheEvent } from "./responseLogger.js";
import { initSearxngCache } from "./searxngSearch.js";
import { runTextOnlyBatch } from "./textBatchProcessor.js";
import {
findSimilarTextModeration,
@@ -70,7 +70,7 @@ export async function runModerationAnalysis(
): Promise<ModerationOutput> {
const { targets, contextBlock, attachments } = input;
initSearxngCache(config.REDIS_URL);
initCacheStore(config.REDIS_URL);
if (!targets.length) throw new Error("No targets provided for analysis");
// ── Phase 1: exact-hash cache (per conversation context) ────────────────
@@ -1,262 +0,0 @@
import Redis from "ioredis";
import { createChildLogger } from "@/shared/logger/index";
import { createAbortControllerWithTimeout } from "@/shared/utils/index";
import { config } from "../../shared/config/config.js";
const log = createChildLogger("searxng-search");
const SEARXNG_BASE_URL = config.SEARXNG_BASE_URL;
const MAX_RESULTS = 3;
const TIMEOUT_MS = 8000;
const CACHE_TTL = 86400; // 24 hours
const CACHE_PREFIX = "searxng:";
let redis: Redis | null = null;
/**
* Exposes the shared SearXNG Redis connection so other modules (e.g. the
* term glossary) reuse the same connection and cache prefix instead of
* opening their own. Returns null when Redis is unavailable.
*/
export function getSearxngRedis(): Redis | null {
return redis;
}
/** Builds a namespaced SearXNG cache key (shared across modules). */
export function makeSearxngCacheKey(namespace: string, key: string): string {
return `${CACHE_PREFIX}${namespace}:${key.toLowerCase().trim()}`;
}
/** Reads a value from the SearXNG Redis cache; null on miss/unavailable. */
export async function searxngCacheGet(key: string): Promise<string | null> {
if (!redis) return null;
try {
return await redis.get(key);
} catch {
return null;
}
}
/** Writes a value to the SearXNG Redis cache, fire-and-forget. */
export function searxngCacheSet(
key: string,
value: string,
ttlSeconds: number,
): void {
if (!redis) return;
redis.setex(key, ttlSeconds, value).catch(() => {
// Cache write failed silently
});
}
/**
* Initialize Redis connection for SearXNG cache.
* Safe to call multiple times — only creates one connection.
*/
export function initSearxngCache(redisUrl: string): void {
if (redis) return;
// Dedicated Redis connection needed because: this connection serves as an
// optional cache for SearXNG web search results with graceful degradation
// when Redis is unavailable (lazyConnect + null-assignment on failure).
// It uses custom retry strategy and must not block or break the main event
// pipeline if the cache is down.
redis = new Redis(redisUrl, {
maxRetriesPerRequest: 3,
retryStrategy(times) {
const delay = Math.min(times * 200, 2000);
return delay;
},
lazyConnect: true,
enableReadyCheck: false,
});
redis.on("error", (err) => {
log.warn({ err: err.message }, "SearXNG Redis cache error");
});
redis.connect().catch(() => {
log.warn("SearXNG Redis cache unavailable — falling back to no-cache");
redis = null;
});
log.info("SearXNG Redis cache initialized");
}
export interface SearxngResult {
title: string;
url: string;
snippet: string;
}
/**
* Search SearXNG for a query and return structured results.
* Uses Redis cache when available — same query within 24h returns cached results.
*
* @param engines Optional comma-separated SearXNG engine list to constrain
* the search (e.g. "wikipedia"). When set, results are cached under a
* separate cache namespace so engine-specific results never collide.
*/
export async function searchSearxng(
query: string,
category: "general" | "news" | "science" = "general",
engines?: string,
timeoutMs: number = TIMEOUT_MS,
): Promise<SearxngResult[]> {
const engineNs = engines ? `eng:${engines}` : "auto";
const cacheKey = makeSearxngCacheKey(`${category}:${engineNs}`, query);
// Try cache first
if (redis) {
try {
const cached = await redis.get(cacheKey);
if (cached) {
log.debug({ query, category, engines }, "SearXNG cache HIT");
return JSON.parse(cached) as SearxngResult[];
}
} catch {
// Cache read failed, continue to API
}
}
// Cache miss — hit SearXNG API
try {
const engineParam = engines
? `&engines=${encodeURIComponent(engines)}`
: "";
const url = `${SEARXNG_BASE_URL}/search?q=${encodeURIComponent(query)}&format=json&language=id&categories=${category}${engineParam}`;
const { controller, clear } = createAbortControllerWithTimeout(timeoutMs);
try {
const response = await fetch(url, {
signal: controller.signal,
headers: {
Accept: "application/json",
"User-Agent":
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
},
});
if (!response.ok) {
log.warn({ status: response.status, query }, "SearXNG search failed");
return [];
}
const data = (await response.json()) as {
results?: Array<{ title?: string; url?: string; content?: string }>;
};
const results = data.results ?? [];
const mapped = results.slice(0, MAX_RESULTS).map((r) => ({
title: r.title ?? "",
url: r.url ?? "",
snippet: (r.content ?? "").slice(0, 500),
}));
// Store in cache (fire and forget — don't block on write)
if (redis) {
redis.setex(cacheKey, CACHE_TTL, JSON.stringify(mapped)).catch(() => {
// Cache write failed silently
});
}
log.debug(
{ query, category, resultCount: mapped.length },
"SearXNG search OK",
);
return mapped;
} finally {
clear();
}
} catch (err) {
log.warn(
{ error: err instanceof Error ? err.message : String(err), query },
"SearXNG search error",
);
return [];
}
}
/**
* Extract meaningful search queries from message content.
* Uses multiple strategies to find terms worth searching.
* Returns up to 3 clean queries.
*/
export function extractSearchQueries(content: string): string[] {
const queries = new Set<string>();
// 1. Quoted phrases (explicit user intent)
const quotedPhrases = content.match(/"([^"]+)"|'([^']+)'/g);
if (quotedPhrases) {
for (const phrase of quotedPhrases) {
const clean = phrase.replace(/["']/g, "").trim();
if (clean.length >= 3) queries.add(clean);
}
}
// 2. "nonton X" pattern — extract the title
const nontonMatch = content.match(
/\b(nonton|tonton|rekomen|cari|search|google)\s+(.+?)(?:\s+(?:anime|kartun|film|movie|series|serial))?\s*[!?.]*$/i,
);
if (nontonMatch) {
const title = nontonMatch[2].trim();
if (title.length >= 2 && title.length <= 80) {
queries.add(title);
}
}
// 3. "X anime/film" pattern — title before category
const titleBeforeCategory = content.match(
/\b(\w[\w\s]{2,40})\s+(?:anime|kartun|film|movie|series|serial)\b/i,
);
if (titleBeforeCategory) {
const title = titleBeforeCategory[1].trim();
if (
title.length >= 3 &&
!/^(yang|yang|sama|dari|untuk|ini|itu|ada)$/i.test(title)
) {
queries.add(title);
}
}
// 4. Standalone proper nouns (2+ words, capitalized) that look like titles
const properNouns = content.match(
/\b([A-Z][a-z]+(?:\s+[A-Z][a-z]+){1,4})\b/g,
);
if (properNouns) {
for (const noun of properNouns) {
// Skip common non-title proper nouns
const skip =
/^(Discord|YouTube|Google|Facebook|Instagram|Twitter|Github|ChatGPT|OpenAI|Claude|Telegram|WhatsApp|TikTok|Netflix|Spotify|Steam|Instagram)$/i;
if (!skip.test(noun) && noun.length >= 5) {
queries.add(noun);
}
}
}
// 5. Terms that suggest research intent
const researchTerms = content.match(
/\b(apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti|meaning|definisi|definition)\s+(.{3,60})/i,
);
if (researchTerms) {
const term = researchTerms[2].trim().replace(/[?!.]+$/, "");
if (term.length >= 3) queries.add(term);
}
return Array.from(queries).slice(0, 3);
}
/**
* Format SearXNG results as XML for LLM context.
*/
export function formatSearchResults(results: SearxngResult[]): string {
if (results.length === 0) return "";
const lines = results.map(
(r) =>
` <result title="${escapeXml(r.title)}">${escapeXml(r.snippet)}</result>`,
);
return `<web_search>\n${lines.join("\n")}\n</web_search>`;
}
function escapeXml(str: string): string {
return str
.replace(/&/g, "&amp;")
.replace(/</g, "&lt;")
.replace(/>/g, "&gt;")
.replace(/"/g, "&quot;");
}
@@ -10,18 +10,18 @@
* wording (false negative on an unknown vulgar/slang term).
*
* Solution: extract candidate "unknown-looking" words from message content,
* look each one up on Wikipedia via SearXNG, and inject the definitions into
* the LLM prompt as a `<term_glossary>` block so verdicts are based on facts
* instead of guesses.
* look each one up on Wikipedia via the Wikipedia REST/Action APIs, and inject
* the definitions into the LLM prompt as a `<term_glossary>` block so verdicts
* are based on facts instead of guesses.
*
* Cost control & persistence:
* - successfully resolved definitions are PERSISTED PERMANENTLY in Postgres
* (`term_glossary_cache`) — definitions rarely change, so a resolved term
* is never searched again; only misses stay ephemeral (Redis/LRU, 1h);
* - in-memory LRU + Redis (shared with the SearXNG cache) sit in front of
* - in-memory LRU + Redis (shared cache store) sit in front of
* the DB as fast read caches, so repeat lookups are effectively free;
* - lookups per batch are bounded (AI_GLOSSARY_MAX_TERMS);
* - live SearXNG calls are rate-limit aware: concurrency 2 + stagger, retry
* - live Wikipedia calls are rate-limit aware: concurrency 2 + stagger, retry
* once on empty results, and misses cached for only 1h so a limiter/
* network blip is not treated as a permanent miss;
* - only results that read like actual definitions are accepted (Wikipedia
@@ -35,17 +35,13 @@ import pLimit from "p-limit";
import { createChildLogger } from "@/shared/logger/index";
import { delay } from "@/shared/utils/index";
import { config } from "../../shared/config/config.js";
import { cacheGet, cacheSet, makeCacheKey } from "./cacheStore.js";
import { escapeXml } from "./moderationBuilders.js";
import {
makeSearxngCacheKey,
searchSearxng,
searxngCacheGet,
searxngCacheSet,
} from "./searxngSearch.js";
import {
getTermDefinitionFromDb,
setTermDefinitionInDb,
} from "./termGlossaryStore.js";
import { wikipediaSummary } from "./wikipediaClient.js";
const log = createChildLogger("term-glossary");
@@ -65,16 +61,13 @@ const MISS_TTL_SECONDS = 60 * 60;
const MISS_TTL_MS = MISS_TTL_SECONDS * 1000;
/** Sentinel stored in caches for "term has no resolvable definition". */
const EMPTY_SENTINEL = "__not_found__";
/** Per-search timeout — keep glossary lookups snappy even on a slow SearXNG. */
const GLOSSARY_SEARCH_TIMEOUT_MS = 5000;
/** Delay before retrying a search that returned zero results. */
const RETRY_DELAY_MS = 350;
/** Max definition snippet length kept in the prompt. */
const MAX_DEFINITION_CHARS = 300;
/**
* SearXNG rate-limits aggressive parallel bursts (returns 200 with empty
* results). Never fire all terms at once — cap live searches at 2 concurrent
* and stagger the start times slightly.
* Wikipedia can be flaky under aggressive parallel bursts. Never fire all
* terms at once — cap live lookups at 2 concurrent and stagger the start times.
*/
const LIVE_SEARCH_CONCURRENCY = 2;
const LIVE_SEARCH_STAGGER_MS = 250;
@@ -90,7 +83,7 @@ const termLru = new LRUCache<string, TermDefinition>({
ttl: 24 * 60 * 60 * 1000,
});
/** Serializes live SearXNG lookups (rate-limit aware) with a small stagger. */
/** Serializes live Wikipedia lookups (rate-limit aware) with a small stagger. */
const liveSearchLimit = pLimit(LIVE_SEARCH_CONCURRENCY);
let lastLiveSearchAt = 0;
async function acquireLiveSlot(): Promise<void> {
@@ -274,58 +267,8 @@ export interface TermDefinition {
sourceUrl: string;
}
/** Definition-like markers for accepting a non-Wikipedia search result. */
const DEF_MARKERS =
/adalah|merupakan|istilah (?:untuk|yang|yg)|artinya|sebutan|berarti|refers? to|known as|also called|short for|a term (?:for|used)|istilah dalam|kata (?:asing|serapan)? ?untuk/i;
/** True when the term appears in the result text (or a 4+ char word in the
* result is part of the term). Lenient — "kafircel" matches a "Kafir"
* article via substring, while a Google-Translate homepage snippet does not. */
function hasTermOverlap(term: string, title: string, snippet: string): boolean {
const termLower = term.toLowerCase();
const text = `${title} ${snippet}`.toLowerCase();
if (text.includes(termLower)) return true;
const words = text.match(/[a-z0-9]{4,}/gi) ?? [];
return words.some((w) => termLower.includes(w));
}
/** Quality gate: is this result good enough to quote as a definition? */
function isUsableDefinition(
r: { title: string; url: string; snippet: string },
term: string,
isWiki: boolean,
): boolean {
const text = `${r.title} ${r.snippet}`;
// Wikipedia disambiguation pages are not definitions
if (/disambiguasi|disambiguation/i.test(text)) return false;
if ((r.snippet ?? "").trim().length < 25) return false;
if (!hasTermOverlap(term, r.title, r.snippet)) return false;
// Wikipedia articles are accepted with just the overlap+length gate;
// everything else must read like an actual definition, not an ad,
// a translate homepage, or a navigation blurb.
if (isWiki) return true;
return DEF_MARKERS.test(r.snippet);
}
/** Picks the best definition from search results, preferring a genuine
* Wikipedia article; otherwise the first result that reads like a
* definition. Returns null when nothing qualifies. */
function pickDefinition(
results: Array<{ title: string; url: string; snippet: string }>,
term: string,
): TermDefinition | null {
const wiki = results.find((r) => /wikipedia\.org/i.test(r.url));
const best = wiki && isUsableDefinition(wiki, term, true) ? wiki : null;
if (!best) {
for (const r of results) {
if (isUsableDefinition(r, term, false)) {
return buildDefinition(r, term);
}
}
return null;
}
return buildDefinition(best, term);
}
/** Per-search timeout — keep glossary lookups snappy even on a slow Wikipedia. */
const GLOSSARY_SEARCH_TIMEOUT_MS = 5000;
function buildDefinition(
best: { title: string; url: string; snippet: string },
@@ -339,7 +282,7 @@ function buildDefinition(
return { term, definition, sourceUrl: best.url };
}
/** Live (network) lookup — runs under the shared SearXNG rate-limit gate. */
/** Live (network) lookup — runs under the shared Wikipedia rate-limit gate. */
async function fetchDefinitionLive(
term: string,
key: string,
@@ -348,31 +291,21 @@ async function fetchDefinitionLive(
return liveSearchLimit(async () => {
await acquireLiveSlot();
try {
let results = await searchSearxng(
key,
"general",
undefined,
GLOSSARY_SEARCH_TIMEOUT_MS,
);
let def = pickDefinition(results, term);
// Zero results is usually the limiter kicking in, not a real miss —
// retry once. Results-but-unusable = genuine miss, no retry.
if (!def && results.length === 0) {
let result = await wikipediaSummary(key, GLOSSARY_SEARCH_TIMEOUT_MS);
let def = result ? buildDefinition(result, term) : null;
// Zero result is usually the limiter/network blip, not a real miss —
// retry once. Result-but-unusable = genuine miss, no retry.
if (!def) {
await delay(RETRY_DELAY_MS);
results = await searchSearxng(
key,
"general",
undefined,
GLOSSARY_SEARCH_TIMEOUT_MS,
);
def = pickDefinition(results, term);
result = await wikipediaSummary(key, GLOSSARY_SEARCH_TIMEOUT_MS);
def = result ? buildDefinition(result, term) : null;
}
if (def) {
// Persist permanently (definitions rarely change) — best-effort,
// then warm the fast caches.
void setTermDefinitionInDb(key, def.definition, def.sourceUrl);
searxngCacheSet(
cacheSet(
cacheKey,
JSON.stringify({
definition: def.definition,
@@ -393,13 +326,13 @@ async function fetchDefinitionLive(
// No definition — cache the miss with a SHORT TTL so a transient
// limiter/network failure is retried on a later batch.
searxngCacheSet(cacheKey, EMPTY_SENTINEL, MISS_TTL_SECONDS);
cacheSet(cacheKey, EMPTY_SENTINEL, MISS_TTL_SECONDS);
termLru.set(key, NOT_FOUND, { ttl: MISS_TTL_MS });
return null;
});
}
/** Resolve one term: LRU → Redis → Postgres (permanent) → live SearXNG
/** Resolve one term: LRU → Redis → Postgres (permanent) → live Wikipedia
* (rate-limited). The fast caches sit in front of the DB; the DB is the
* source of truth for successfully resolved definitions. */
async function resolveTerm(term: string): Promise<TermDefinition | null> {
@@ -412,8 +345,8 @@ async function resolveTerm(term: string): Promise<TermDefinition | null> {
// 2. Redis — shared across processes/workers. A miss sentinel here is NOT
// a definitive answer: it may predate a permanent DB entry written by
// another process, so we keep going and let the DB decide.
const cacheKey = makeSearxngCacheKey("def", key);
const cached = await searxngCacheGet(cacheKey);
const cacheKey = makeCacheKey("def", key);
const cached = await cacheGet(cacheKey);
let redisMiss = false;
if (cached !== null) {
if (cached === EMPTY_SENTINEL) {
@@ -449,7 +382,7 @@ async function resolveTerm(term: string): Promise<TermDefinition | null> {
sourceUrl: dbDef.sourceUrl,
};
termLru.set(key, def);
searxngCacheSet(
cacheSet(
cacheKey,
JSON.stringify({ definition: def.definition, sourceUrl: def.sourceUrl }),
DEF_TTL_SECONDS,
@@ -1,7 +1,7 @@
/**
* textBatchProcessor.ts
*
* Processes text-only moderation batches fetches URL content, runs SearXNG
* Processes text-only moderation batches fetches URL content, runs Wikipedia
* searches, deduplicates short messages, splits into sub-batches, and calls
* the LLM for analysis. Extracted from moderationOrchestrator.ts.
*/
@@ -28,15 +28,15 @@ import {
} from "./moderationBuilders.js";
import { buildSystemPrompt as buildSystemPromptModular } from "./moderationPrompt.js";
import { logModerationAnalysis } from "./responseLogger.js";
import {
extractSearchQueries,
formatSearchResults,
searchSearxng,
} from "./searxngSearch.js";
import { buildTermGlossaryBlock } from "./termGlossary.js";
import { getRecentCorrectedModerations } from "./textCacheStore.js";
import { extractUrlsFromText, fetchUrlSafely } from "./urlFetcher.js";
import type { MessageImagePart } from "./visionAnalyzer.js";
import {
extractSearchQueries,
formatSearchResults,
wikipediaSearch,
} from "./wikipediaClient.js";
const log = createChildLogger("textBatchProcessor");
@@ -117,7 +117,7 @@ export async function runTextOnlyBatch(
return { text: textMap, image: imageMap, title: titleMap };
})();
const searxngPromise = (async () => {
const webSearchPromise = (async () => {
const queries = new Set<string>();
for (const msg of targets) {
for (const q of extractSearchQueries(msg.edited_content ?? msg.content))
@@ -126,7 +126,7 @@ export async function runTextOnlyBatch(
if (queries.size === 0) return new Map<string, string>();
const queryArr = Array.from(queries).slice(0, 3);
const results = await Promise.allSettled(
queryArr.map((q) => searchSearxng(q)),
queryArr.map((q) => wikipediaSearch(q)),
);
const map = new Map<string, string>();
for (let i = 0; i < queryArr.length; i++) {
@@ -138,15 +138,13 @@ export async function runTextOnlyBatch(
})();
// Term glossary — per-word Wikipedia lookups for words the LLM may not
// know (slang, jargon, regional language). Cached in Redis + in-memory, so
// repeat terms resolve instantly and only genuinely new words hit SearXNG.
const glossaryPromise = buildTermGlossaryBlock(
targets.map((msg) => getAnalysisContent(msg)),
).catch(() => "");
const [urlFetchMaps, searxngResults, glossaryBlock] = await Promise.all([
const [urlFetchMaps, webSearchResults, glossaryBlock] = await Promise.all([
urlFetchPromise,
searxngPromise,
webSearchPromise,
glossaryPromise,
]);
const urlFetchMap = urlFetchMaps.text;
@@ -308,9 +306,9 @@ export async function runTextOnlyBatch(
)
).join("\n");
const searxngBlock =
searxngResults.size > 0
? `<web_searches>\n${Array.from(searxngResults.entries())
const webSearchBlock =
webSearchResults.size > 0
? `<web_searches>\n${Array.from(webSearchResults.entries())
.map(
([q, xml]) =>
` <search_query query="${escapeXml(q)}">\n${xml} </search_query>`,
@@ -323,7 +321,7 @@ export async function runTextOnlyBatch(
// profile descriptions are intentionally omitted (see above).
const userBlocks = [
contextBlock?.trimEnd() ?? "",
searxngBlock,
webSearchBlock,
glossaryBlock,
`<messages_to_analyze>\n${messagesBlock}\n</messages_to_analyze>`,
].filter((b) => b.trim().length > 0);
@@ -76,13 +76,13 @@ import {
buildStickerTextOnlyWarning,
buildStickerVisionPrompt,
} from "./moderationPrompt.js";
import { buildTermGlossaryBlock } from "./termGlossary.js";
import { extractUrlsFromText } from "./urlFetcher.js";
import {
extractSearchQueries,
formatSearchResults,
searchSearxng,
} from "./searxngSearch.js";
import { buildTermGlossaryBlock } from "./termGlossary.js";
import { extractUrlsFromText } from "./urlFetcher.js";
wikipediaSearch,
} from "./wikipediaClient.js";
// ---------------------------------------------------------------------------
// Types
@@ -354,12 +354,12 @@ export async function prepareMediaMessage(
),
);
// SearXNG
let searxngXml = "";
// Wikipedia web search (context enrichment)
let webSearchXml = "";
const queries = extractSearchQueries(content);
if (queries.length > 0) {
const results = await Promise.allSettled(
queries.map((q) => searchSearxng(q)),
queries.map((q) => wikipediaSearch(q)),
);
const parts: string[] = [];
for (let i = 0; i < results.length; i++) {
@@ -368,7 +368,7 @@ export async function prepareMediaMessage(
parts.push(formatSearchResults(r.value));
}
if (parts.length > 0)
searxngXml = `\n<web_searches>\n${parts.join("\n")}\n</web_searches>`;
webSearchXml = `\n<web_searches>\n${parts.join("\n")}\n</web_searches>`;
}
// Term glossary — cached per-word Wikipedia definitions for words the LLM
@@ -403,6 +403,6 @@ export async function prepareMediaMessage(
// still tracked in the DB for enforcement, just not shown to the LLM.
const isBot = resolveIsBot(target);
const isEdited = resolveIsEdited(target);
const messageBlock = `<message id="${escapeXml(target.id)}" user="${escapeXml(resolveDisplayName(target))}" time="${new Date(target.created_at).toISOString()}"${isBot ? ` bot="true"` : ""}${isEdited ? ` edited="true"` : ""}>\n ${refXml ? `\n ${refXml}` : ""}\n <content>${escapeXml(truncateForAi(content))}</content>${mediaContext ? ` ${escapeXml(mediaContext)}` : ""}${webContext}${mediaAnalysisContext}${searxngXml}${glossaryCtx}\n</message>`;
const messageBlock = `<message id="${escapeXml(target.id)}" user="${escapeXml(resolveDisplayName(target))}" time="${new Date(target.created_at).toISOString()}"${isBot ? ` bot="true"` : ""}${isEdited ? ` edited="true"` : ""}>\n ${refXml ? `\n ${refXml}` : ""}\n <content>${escapeXml(truncateForAi(content))}</content>${mediaContext ? ` ${escapeXml(mediaContext)}` : ""}${webContext}${mediaAnalysisContext}${webSearchXml}${glossaryCtx}\n</message>`;
return { targetId, messageBlock };
}
@@ -0,0 +1,260 @@
/**
* wikipediaClient.ts
*
* Wikipedia adapter for AI analysis context enrichment.
*
* Replaces the old SearXNG web-search dependency (removed). Instead of a
* meta-search instance, we talk to the public Wikipedia REST + Action APIs
* directly with native `fetch` no extra npm dependency, full control, and
* a stable, well-documented endpoint.
*
* Layer exposed to the moderation pipeline:
* WikipediaClient
* search() list(query) (Action API: opensearch-like)
* getSummary() summary(title) (REST summary endpoint)
* (page content) page(title) [reserved]
*
* The functions below are thin wrappers matching the old consumer surface so
* call sites change as little as possible.
*/
import { createChildLogger } from "@/shared/logger/index";
import { createAbortControllerWithTimeout } from "@/shared/utils/index";
import { config } from "../../shared/config/config.js";
const log = createChildLogger("wikipedia-client");
const WIKIPEDIA_LANG = config.WIKIPEDIA_LANG.toLowerCase();
const MAX_RESULTS = 3;
const DEFAULT_TIMEOUT_MS = config.WIKIPEDIA_TIMEOUT_MS;
/** Canonical article URL for a title in the active wiki language. */
export function wikipediaPageUrl(title: string): string {
return `https://${WIKIPEDIA_LANG}.wikipedia.org/wiki/${encodeURIComponent(
title.trim().replace(/ /g, "_"),
)}`;
}
export interface SearchResult {
title: string;
url: string;
snippet: string;
}
function buildUserAgent(): string {
return "GMWBeta/1.0 (https://github.com/asepharyana; Discord moderation bot)";
}
function stripHtml(snippet: string): string {
return snippet
.replace(/<[^>]+>/g, "")
.replace(/&quot;/g, '"')
.replace(/&amp;/g, "&")
.replace(/&#39;/g, "'")
.replace(/&lt;/g, "<")
.replace(/&gt;/g, ">")
.replace(/\s+/g, " ")
.trim();
}
/**
* Search Wikipedia for a query and return up to MAX_RESULTS structured hits.
* Uses the Action API `list=search` (srsearch) which is stable and returns
* title + HTML snippet. Graceful: returns [] on any failure.
*/
export async function wikipediaSearch(
query: string,
timeoutMs: number = DEFAULT_TIMEOUT_MS,
): Promise<SearchResult[]> {
const q = query.trim();
if (!q) return [];
const params = new URLSearchParams({
action: "query",
list: "search",
srsearch: q,
srlimit: String(MAX_RESULTS),
format: "json",
origin: "*",
});
const { controller, clear } = createAbortControllerWithTimeout(timeoutMs);
try {
const res = await fetch(
`https://${WIKIPEDIA_LANG}.wikipedia.org/w/api.php?${params.toString()}`,
{
signal: controller.signal,
headers: {
Accept: "application/json",
"User-Agent": buildUserAgent(),
},
},
);
if (!res.ok) {
log.warn({ status: res.status, query: q }, "Wikipedia search failed");
return [];
}
const data = (await res.json()) as {
query?: { search?: Array<{ title: string; snippet?: string }> };
};
const hits = data.query?.search ?? [];
const mapped = hits.slice(0, MAX_RESULTS).map((h) => ({
title: h.title,
url: wikipediaPageUrl(h.title),
snippet: stripHtml(h.snippet ?? "").slice(0, 500),
}));
log.debug({ query: q, resultCount: mapped.length }, "Wikipedia search OK");
return mapped;
} catch (err) {
log.warn(
{ error: err instanceof Error ? err.message : String(err), query: q },
"Wikipedia search error",
);
return [];
} finally {
clear();
}
}
/**
* Fetch the lead summary of a specific Wikipedia article via the REST
* summary endpoint. Returns null when the article is missing or the request
* fails. Useful for the term glossary's direct lookups.
*/
export async function wikipediaSummary(
title: string,
timeoutMs: number = DEFAULT_TIMEOUT_MS,
): Promise<SearchResult | null> {
const t = title.trim();
if (!t) return null;
const { controller, clear } = createAbortControllerWithTimeout(timeoutMs);
try {
const res = await fetch(
`https://${WIKIPEDIA_LANG}.wikipedia.org/api/rest_v1/page/summary/${encodeURIComponent(
t.replace(/ /g, "_"),
)}`,
{
signal: controller.signal,
headers: {
Accept: "application/json",
"User-Agent": buildUserAgent(),
},
},
);
if (!res.ok) return null;
const data = (await res.json()) as {
title?: string;
extract?: string;
content_urls?: { desktop?: { page?: string } };
};
if (!data.extract) return null;
return {
title: data.title ?? t,
url: data.content_urls?.desktop?.page ?? wikipediaPageUrl(t),
snippet: data.extract.slice(0, 500),
};
} catch (err) {
log.warn(
{ error: err instanceof Error ? err.message : String(err), title: t },
"Wikipedia summary error",
);
return null;
} finally {
clear();
}
}
/**
* Extract meaningful search queries from message content.
* Uses multiple strategies to find terms worth searching.
* Returns up to 3 clean queries.
*/
export function extractSearchQueries(content: string): string[] {
const queries = new Set<string>();
// 1. Quoted phrases (explicit user intent)
const quotedPhrases = content.match(/"([^"]+)"|'([^']+)'/g);
if (quotedPhrases) {
for (const phrase of quotedPhrases) {
const clean = phrase.replace(/["']/g, "").trim();
if (clean.length >= 3) queries.add(clean);
}
}
// 2. "nonton X" pattern — extract the title
const nontonMatch = content.match(
/\b(nonton|tonton|rekomen|cari|search|google)\s+(.+?)(?:\s+(?:anime|kartun|film|movie|series|serial))?\s*[!?.]*$/i,
);
if (nontonMatch) {
const title = nontonMatch[2].trim();
if (title.length >= 2 && title.length <= 80) {
queries.add(title);
}
}
// 3. "X anime/film" pattern — title before category
const titleBeforeCategory = content.match(
/\b(\w[\w\s]{2,40})\s+(?:anime|kartun|film|movie|series|serial)\b/i,
);
if (titleBeforeCategory) {
const title = titleBeforeCategory[1].trim();
if (
title.length >= 3 &&
!/^(yang|yang|sama|dari|untuk|ini|itu|ada)$/i.test(title)
) {
queries.add(title);
}
}
// 4. Standalone proper nouns (2+ words, capitalized) that look like titles
const properNouns = content.match(
/\b([A-Z][a-z]+(?:\s+[A-Z][a-z]+){1,4})\b/g,
);
if (properNouns) {
for (const noun of properNouns) {
// Skip common non-title proper nouns
const skip =
/^(Discord|YouTube|Google|Facebook|Instagram|Twitter|Github|ChatGPT|OpenAI|Claude|Telegram|WhatsApp|TikTok|Netflix|Spotify|Steam|Instagram)$/i;
if (!skip.test(noun) && noun.length >= 5) {
queries.add(noun);
}
}
}
// 5. Terms that suggest research intent
const researchTerms = content.match(
/\b(apa\s+(?:itu|sih)|what\s+is|siapa\s+itu|who\s+is|arti|meaning|definisi|definition)\s+(.{3,60})/i,
);
if (researchTerms) {
const term = researchTerms[2].trim().replace(/[?!.]+$/, "");
if (term.length >= 3) queries.add(term);
}
return Array.from(queries).slice(0, 3);
}
/**
* Format Wikipedia results as XML for LLM context.
*/
export function formatSearchResults(results: SearchResult[]): string {
if (results.length === 0) return "";
const lines = results.map(
(r) =>
` <result title="${escapeXml(r.title)}">${escapeXml(r.snippet)}</result>`,
);
return `<web_search>\n${lines.join("\n")}\n</web_search>`;
}
function escapeXml(str: string): string {
return str
.replace(/&/g, "&amp;")
.replace(/</g, "&lt;")
.replace(/>/g, "&gt;")
.replace(/"/g, "&quot;");
}
@@ -104,10 +104,12 @@ export const configSchema = z
// ── Redis ────────────────────────────────────────────────────────────
REDIS_URL: z.string().default("redis://localhost:6379"),
// ── SearXNG ───────────────────────────────────────────────────────────
// Instance for web search + term glossary lookups. Override when the
// default instance is down/rate-limited.
SEARXNG_BASE_URL: z.string().url().default("https://searxng.imrnes.team"),
// ── Wikipedia (web-search / glossary source) ─────────────────────────
// Native fetch to Wikipedia REST + Action APIs — no SearXNG dependency.
// Language for summaries/search (e.g. "id", "en").
WIKIPEDIA_LANG: z.string().min(1).default("id"),
// Per-request timeout (ms) for Wikipedia API calls.
WIKIPEDIA_TIMEOUT_MS: z.coerce.number().positive().default(8000),
// ── Voice PCM WebSocket (direct gateway→backend, bypasses Redis) ────
VOICE_PCM_WS_ENABLED: z
.string()