2026-06-27 15:53:42 +07:00
|
|
|
/**
|
|
|
|
|
* LRU Response Cache with TTL.
|
|
|
|
|
*
|
|
|
|
|
* Designed for LLM proxy responses — reduces bandwidth and upstream
|
|
|
|
|
* round-trips for repeated prompts (common during development, retries,
|
|
|
|
|
* or shared prefixes).
|
|
|
|
|
*
|
|
|
|
|
* Strategy:
|
2026-06-27 16:56:21 +07:00
|
|
|
* - LRU eviction via Map insertion order (O(1) reorder on access)
|
|
|
|
|
* - Configurable TTL per entry (default 300s — balances freshness vs hit rate)
|
2026-06-27 15:53:42 +07:00
|
|
|
* - Cache key = hash of (model + sorted messages + stream flag)
|
2026-06-27 16:56:21 +07:00
|
|
|
* - Non-streaming responses are cached; streaming not cached (would need
|
|
|
|
|
* full body buffering)
|
|
|
|
|
* - Model allowlist via CACHE_MODELS envvar (comma-separated prefixes)
|
|
|
|
|
* - Basic hit/miss stats exported for observability
|
2026-06-27 15:53:42 +07:00
|
|
|
*
|
2026-06-27 16:56:21 +07:00
|
|
|
* Performance: Uses Map.delete+set instead of a hand-rolled linked list,
|
|
|
|
|
* giving O(1) reorder on access vs O(n) scan in the previous implementation.
|
2026-06-27 15:53:42 +07:00
|
|
|
*/
|
|
|
|
|
|
|
|
|
|
// ─── Cache entry ──────────────────────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
interface CacheEntry {
|
|
|
|
|
body: string;
|
|
|
|
|
status: number;
|
|
|
|
|
headers: Record<string, string>;
|
|
|
|
|
createdAt: number;
|
|
|
|
|
expiresAt: number;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
// ─── Cache stats ──────────────────────────────────────────────────────────
|
2026-06-27 15:53:42 +07:00
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
export interface CacheStats {
|
|
|
|
|
hits: number;
|
|
|
|
|
misses: number;
|
|
|
|
|
size: number;
|
|
|
|
|
maxSize: number;
|
|
|
|
|
hitRate: number;
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ─── ResponseCache class ──────────────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
export class ResponseCache {
|
2026-06-27 16:56:21 +07:00
|
|
|
/** Map preserves insertion order — used as LRU ordering */
|
2026-06-27 15:53:42 +07:00
|
|
|
private readonly map = new Map<string, CacheEntry>();
|
|
|
|
|
private readonly maxSize: number;
|
|
|
|
|
private readonly defaultTtlMs: number;
|
2026-06-27 16:56:21 +07:00
|
|
|
private readonly modelAllowlist: RegExp[];
|
|
|
|
|
private hits = 0;
|
|
|
|
|
private misses = 0;
|
2026-06-27 15:53:42 +07:00
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
constructor(opts?: {
|
|
|
|
|
maxSize?: number;
|
|
|
|
|
defaultTtlMs?: number;
|
|
|
|
|
modelAllowlist?: RegExp[];
|
|
|
|
|
}) {
|
2026-06-27 15:53:42 +07:00
|
|
|
this.maxSize = opts?.maxSize ?? 500;
|
2026-06-27 16:56:21 +07:00
|
|
|
this.defaultTtlMs = opts?.defaultTtlMs ?? 300_000; // 300 seconds (5 min)
|
|
|
|
|
this.modelAllowlist = opts?.modelAllowlist ?? [];
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ── Public API ──────────────────────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
/** Build a deterministic cache key from an LLM request. */
|
|
|
|
|
static buildKey(model: string, messages: unknown, stream: boolean): string {
|
2026-06-27 17:46:51 +07:00
|
|
|
const stable = stableStringifyCached(messages);
|
2026-06-27 15:53:42 +07:00
|
|
|
const raw = `${model}|${stream}|${stable}`;
|
|
|
|
|
return simpleHash(raw);
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
/** Check if this model should be cached. */
|
|
|
|
|
shouldCacheModel(model: string): boolean {
|
|
|
|
|
if (this.modelAllowlist.length === 0) return true;
|
|
|
|
|
return this.modelAllowlist.some((re) => re.test(model));
|
|
|
|
|
}
|
2026-06-27 15:53:42 +07:00
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
/** Retrieve a cached response. Returns null if missing or expired. */
|
|
|
|
|
get(
|
|
|
|
|
key: string,
|
|
|
|
|
): { body: string; status: number; headers: Record<string, string> } | null {
|
|
|
|
|
if (!this.map.has(key)) {
|
|
|
|
|
this.misses++;
|
2026-06-27 15:53:42 +07:00
|
|
|
return null;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
const entry = this.map.get(key)!;
|
|
|
|
|
|
|
|
|
|
// Expired — evict and return null
|
|
|
|
|
if (Date.now() > entry.expiresAt) {
|
|
|
|
|
this.map.delete(key);
|
|
|
|
|
this.misses++;
|
|
|
|
|
return null;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Move to end (most recently used) — O(1) in Map
|
|
|
|
|
this.map.delete(key);
|
|
|
|
|
this.map.set(key, entry);
|
|
|
|
|
this.hits++;
|
2026-06-27 15:53:42 +07:00
|
|
|
return { body: entry.body, status: entry.status, headers: entry.headers };
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Store a response in the cache. */
|
|
|
|
|
set(
|
|
|
|
|
key: string,
|
|
|
|
|
body: string,
|
|
|
|
|
status: number,
|
|
|
|
|
headers: Record<string, string>,
|
|
|
|
|
ttlMs?: number,
|
|
|
|
|
): void {
|
|
|
|
|
// Enforce max size before inserting
|
|
|
|
|
if (this.map.size >= this.maxSize) {
|
|
|
|
|
this.evictLRU();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
const now = Date.now();
|
|
|
|
|
const ttl = ttlMs ?? this.defaultTtlMs;
|
|
|
|
|
|
|
|
|
|
const entry: CacheEntry = {
|
|
|
|
|
body,
|
|
|
|
|
status,
|
|
|
|
|
headers,
|
|
|
|
|
createdAt: now,
|
|
|
|
|
expiresAt: now + ttl,
|
|
|
|
|
};
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
// Set as most recently used (last in iteration order)
|
|
|
|
|
this.map.delete(key);
|
2026-06-27 15:53:42 +07:00
|
|
|
this.map.set(key, entry);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Delete a specific key. */
|
|
|
|
|
delete(key: string): void {
|
|
|
|
|
this.map.delete(key);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Clear all entries. */
|
|
|
|
|
clear(): void {
|
|
|
|
|
this.map.clear();
|
2026-06-27 16:56:21 +07:00
|
|
|
this.hits = 0;
|
|
|
|
|
this.misses = 0;
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Current number of entries. */
|
|
|
|
|
get size(): number {
|
|
|
|
|
return this.map.size;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Sweep expired entries. Call periodically if desired. */
|
|
|
|
|
sweep(): number {
|
|
|
|
|
const now = Date.now();
|
|
|
|
|
let removed = 0;
|
|
|
|
|
for (const [key, entry] of this.map) {
|
|
|
|
|
if (now > entry.expiresAt) {
|
2026-06-27 16:56:21 +07:00
|
|
|
this.map.delete(key);
|
2026-06-27 15:53:42 +07:00
|
|
|
removed++;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return removed;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
/** Return hit/miss stats and reset counters. */
|
|
|
|
|
stats(): CacheStats {
|
|
|
|
|
const total = this.hits + this.misses;
|
|
|
|
|
return {
|
|
|
|
|
hits: this.hits,
|
|
|
|
|
misses: this.misses,
|
|
|
|
|
size: this.map.size,
|
|
|
|
|
maxSize: this.maxSize,
|
|
|
|
|
hitRate: total > 0 ? this.hits / total : 0,
|
|
|
|
|
};
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
/** Reset hit/miss counters. */
|
|
|
|
|
resetStats(): void {
|
|
|
|
|
this.hits = 0;
|
|
|
|
|
this.misses = 0;
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
// ── Internals ───────────────────────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
/** Evict the least recently used entry (first in insertion order). */
|
2026-06-27 15:53:42 +07:00
|
|
|
private evictLRU(): void {
|
2026-06-27 16:56:21 +07:00
|
|
|
const lruKey = this.map.keys().next().value;
|
|
|
|
|
if (lruKey !== undefined) {
|
|
|
|
|
this.map.delete(lruKey);
|
|
|
|
|
}
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ─── Helpers ──────────────────────────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* JSON.stringify replacer that sorts object keys for stable hashing.
|
|
|
|
|
*/
|
|
|
|
|
function stableStringifyReplacer(_key: string, value: unknown): unknown {
|
|
|
|
|
if (value !== null && typeof value === "object" && !Array.isArray(value)) {
|
|
|
|
|
const sorted: Record<string, unknown> = {};
|
|
|
|
|
for (const k of Object.keys(value).sort()) {
|
|
|
|
|
sorted[k] = (value as Record<string, unknown>)[k];
|
|
|
|
|
}
|
|
|
|
|
return sorted;
|
|
|
|
|
}
|
|
|
|
|
return value;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-27 17:46:51 +07:00
|
|
|
/**
|
|
|
|
|
* WeakMap cache for stableStringify output — avoids re-stringifying the same
|
|
|
|
|
* request body across retries or repeated identical requests within the same
|
|
|
|
|
* process lifetime. Keyed by the top-level messages reference; entries are
|
|
|
|
|
* collected once the request body is GC'd.
|
|
|
|
|
*/
|
|
|
|
|
const _stableStringifyCache = new WeakMap<object, string>();
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Memoized stable JSON.stringify for buildKey. Re-using the same `messages`
|
|
|
|
|
* reference across retries hits the WeakMap and skips the recursive sort.
|
|
|
|
|
* Falls back to direct stringify when the value isn't a referenceable object.
|
|
|
|
|
*/
|
|
|
|
|
function stableStringifyCached(value: unknown): string {
|
|
|
|
|
if (value && typeof value === "object") {
|
|
|
|
|
const hit = _stableStringifyCache.get(value as object);
|
|
|
|
|
if (hit !== undefined) return hit;
|
|
|
|
|
const s = JSON.stringify(value, stableStringifyReplacer);
|
|
|
|
|
_stableStringifyCache.set(value as object, s);
|
|
|
|
|
return s;
|
|
|
|
|
}
|
|
|
|
|
return JSON.stringify(value, stableStringifyReplacer);
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-27 15:53:42 +07:00
|
|
|
/**
|
|
|
|
|
* Simple, fast non-cryptographic hash (djb2 variant).
|
|
|
|
|
* Collisions are theoretically possible but extremely unlikely for
|
|
|
|
|
* cache-key use — a collision would serve a wrong cached response,
|
|
|
|
|
* which is bounded by TTL.
|
|
|
|
|
*/
|
|
|
|
|
function simpleHash(input: string): string {
|
|
|
|
|
let hash = 5381;
|
|
|
|
|
for (let i = 0; i < input.length; i++) {
|
|
|
|
|
hash = ((hash << 5) + hash + input.charCodeAt(i)) & 0xffffffff;
|
|
|
|
|
}
|
|
|
|
|
return hash.toString(36);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ─── Module-level singleton (lazy) ────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
let _instance: ResponseCache | null = null;
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Get or create the shared ResponseCache instance.
|
|
|
|
|
* Configured via environment variables:
|
2026-06-27 16:56:21 +07:00
|
|
|
* CACHE_TTL — TTL in ms (0 = disabled, default 300000 = 5 min)
|
2026-06-27 15:53:42 +07:00
|
|
|
* CACHE_MAX_SIZE — max entries (default 500)
|
2026-06-27 16:56:21 +07:00
|
|
|
* CACHE_MODELS — comma-separated model prefixes to cache
|
|
|
|
|
* (e.g. "deepseek,minimax,kimi" — empty = cache all)
|
2026-06-27 15:53:42 +07:00
|
|
|
*/
|
|
|
|
|
export function getResponseCache(): ResponseCache | null {
|
2026-06-27 16:56:21 +07:00
|
|
|
const ttl = Number(process.env.CACHE_TTL ?? 300000);
|
2026-06-27 15:53:42 +07:00
|
|
|
if (ttl <= 0) return null; // explicitly disabled
|
|
|
|
|
|
|
|
|
|
if (!_instance) {
|
|
|
|
|
const maxSize = Number(process.env.CACHE_MAX_SIZE ?? 500);
|
2026-06-27 16:56:21 +07:00
|
|
|
const allowlistRaw = (process.env.CACHE_MODELS ?? "").trim();
|
|
|
|
|
const allowlist: RegExp[] = allowlistRaw
|
|
|
|
|
? allowlistRaw.split(",").map((s) => new RegExp(s.trim()))
|
|
|
|
|
: [];
|
|
|
|
|
_instance = new ResponseCache({
|
|
|
|
|
maxSize,
|
|
|
|
|
defaultTtlMs: ttl,
|
|
|
|
|
modelAllowlist: allowlist,
|
|
|
|
|
});
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
return _instance;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-27 16:56:21 +07:00
|
|
|
// ─── DSML model guard ────────────────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
/** Comma-separated model prefixes that can produce DSML. */
|
|
|
|
|
const DSML_MODELS_DEFAULT = "deepseek,codestral";
|
|
|
|
|
|
|
|
|
|
/** Parse DSML_MODELS from env, cached after first call. */
|
|
|
|
|
let _dsmlPatterns: RegExp[] | null = null;
|
|
|
|
|
|
|
|
|
|
function getDSMLPatterns(): RegExp[] {
|
|
|
|
|
if (!_dsmlPatterns) {
|
|
|
|
|
const raw = (process.env.DSML_MODELS ?? DSML_MODELS_DEFAULT).trim();
|
|
|
|
|
_dsmlPatterns = raw
|
|
|
|
|
.split(",")
|
|
|
|
|
.map((s) => s.trim())
|
|
|
|
|
.filter(Boolean)
|
|
|
|
|
.map((s) => new RegExp(s, "i"));
|
|
|
|
|
}
|
|
|
|
|
return _dsmlPatterns;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Check if DSML detection is globally enabled AND if this model is known
|
|
|
|
|
* to produce DSML output.
|
|
|
|
|
*
|
|
|
|
|
* DSML is a DeepSeek-specific markup format. For all other models (OpenAI,
|
|
|
|
|
* Anthropic, CastAI, etc.) there is no need to scan every SSE chunk.
|
|
|
|
|
*
|
|
|
|
|
* Configured via:
|
|
|
|
|
* DSML_DETECTION — global on/off (default "true")
|
|
|
|
|
* DSML_MODELS — comma-separated model prefixes (default "deepseek,codestral")
|
|
|
|
|
*/
|
|
|
|
|
export function isDSMLDetectionEnabled(model?: string): boolean {
|
|
|
|
|
const globalEnabled = process.env.DSML_DETECTION ?? "true";
|
|
|
|
|
if (globalEnabled !== "true" && globalEnabled !== "1") return false;
|
|
|
|
|
if (!model) return true; // backward compat for non-model callers
|
|
|
|
|
|
|
|
|
|
return getDSMLPatterns().some((re) => re.test(model));
|
2026-06-27 15:53:42 +07:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Check if stream passthrough mode is enabled (env toggle, default true). */
|
|
|
|
|
export function isStreamPassthroughEnabled(): boolean {
|
|
|
|
|
const val = process.env.STREAM_PASSTHROUGH ?? "true";
|
|
|
|
|
return val === "true" || val === "1";
|
|
|
|
|
}
|