perf(ai-moderation): speed up analysis queue (ramai + sepi)
- Parallelize per-user reputation/profile fetches in textBatchProcessor (was a serial ~2N DB/Redis round-trip loop per sub-batch; now Promise.all over unique users). Cuts per-batch latency, biggest win on small/quiet batches. - Make the LLM concurrency semaphore dynamic (cached per config value) instead of frozen at import time, so AI_LLM_MAX_CONCURRENT is tunable without code change and reflects current config. - Bump AI_LLM_MAX_CONCURRENT default 5 -> 8 (gemini-flash-lite is cheap; helps throughput when busy). - Lower AI_ANALYSIS_DEBOUNCE_MS 500 -> 250 (snappier first-message analysis when quiet). - Lower AI_ANALYSIS_RECOVERY_INTERVAL_MS 15000 -> 10000 (stuck/errored messages re-analyze sooner). tsc, biome, vitest (129) all clean.
This commit is contained in:
@@ -18,7 +18,20 @@ const log = createChildLogger("llm-client");
|
||||
// Concurrency limiter for LLM API calls (inlined from concurrencyLimiter.ts)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const llmSemaphore = pLimit(config.AI_LLM_MAX_CONCURRENT ?? 5);
|
||||
// The limiter is cached per configured concurrency value so it can be tuned
|
||||
// (env / BWS) without a code change and always reflects the current config —
|
||||
// a module-level `pLimit(config.X)` would freeze the cap at import time.
|
||||
let llmSemaphore = pLimit(config.AI_LLM_MAX_CONCURRENT ?? 5);
|
||||
let llmSemaphoreLimit = config.AI_LLM_MAX_CONCURRENT ?? 5;
|
||||
|
||||
function getLlmSemaphore() {
|
||||
const wanted = config.AI_LLM_MAX_CONCURRENT ?? 5;
|
||||
if (wanted !== llmSemaphoreLimit) {
|
||||
llmSemaphore = pLimit(wanted);
|
||||
llmSemaphoreLimit = wanted;
|
||||
}
|
||||
return llmSemaphore;
|
||||
}
|
||||
|
||||
let activeCount = 0;
|
||||
let pendingCount = 0;
|
||||
@@ -30,7 +43,7 @@ export async function withLlmConcurrency<T>(fn: () => Promise<T>): Promise<T> {
|
||||
"Queuing LLM request",
|
||||
);
|
||||
|
||||
return llmSemaphore(async () => {
|
||||
return getLlmSemaphore()(async () => {
|
||||
pendingCount--;
|
||||
activeCount++;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user